From 666feef2a97d30b14b2ac4e1d7bc13614f5aff38 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:27:41 +0530 Subject: [PATCH 01/50] Add chat completion support for websearch --- .../websearch_interception/handler.py | 265 +++++++++++++++++- 1 file changed, 258 insertions(+), 7 deletions(-) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 5d36b760afb..1e109dc9e39 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -48,7 +48,8 @@ class WebSearchInterceptionLogger(CustomLogger): Args: enabled_providers: List of LLM providers to enable interception for. Use LlmProviders enum values (e.g., [LlmProviders.BEDROCK]) - Default: [LlmProviders.BEDROCK] + If None or empty list, enables for ALL providers. + Default: None (all providers enabled) search_tool_name: Name of search tool configured in router's search_tools. If None, will attempt to use first available search tool. """ @@ -183,10 +184,10 @@ class WebSearchInterceptionLogger(CustomLogger): verbose_logger.debug( f"WebSearchInterception: Pre-request hook called" f" - custom_llm_provider={custom_llm_provider}" - f" - enabled_providers={self.enabled_providers}" + f" - enabled_providers={self.enabled_providers or 'ALL'}" ) - if custom_llm_provider not in self.enabled_providers: + if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers: verbose_logger.debug( f"WebSearchInterception: Skipping - provider {custom_llm_provider} not in {self.enabled_providers}" ) @@ -245,7 +246,12 @@ class WebSearchInterceptionLogger(CustomLogger): custom_llm_provider: str, kwargs: Dict, ) -> Tuple[bool, Dict]: - """Check if WebSearch tool interception is needed""" + """ + Check if WebSearch tool interception is needed for Anthropic Messages API. + + This is the legacy method for Anthropic-style responses. + For chat completions, use async_should_run_chat_completion_agentic_loop instead. + """ verbose_logger.debug(f"WebSearchInterception: Hook called! provider={custom_llm_provider}, stream={stream}") verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}") @@ -253,7 +259,7 @@ class WebSearchInterceptionLogger(CustomLogger): # Check if provider should be intercepted # Note: custom_llm_provider is already normalized by get_llm_provider() # (e.g., "bedrock/invoke/..." -> "bedrock") - if custom_llm_provider not in self.enabled_providers: + if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers: verbose_logger.debug( f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})" ) @@ -267,10 +273,11 @@ class WebSearchInterceptionLogger(CustomLogger): ) return False, {} - # Detect WebSearch tool_use in response + # Detect WebSearch tool_use in response (Anthropic format) should_intercept, tool_calls = WebSearchTransformation.transform_request( response=response, stream=stream, + response_format="anthropic", ) if not should_intercept: @@ -288,6 +295,67 @@ class WebSearchInterceptionLogger(CustomLogger): "tool_calls": tool_calls, "tool_type": "websearch", "provider": custom_llm_provider, + "response_format": "anthropic", + } + return True, tools_dict + + async def async_should_run_chat_completion_agentic_loop( + self, + response: Any, + model: str, + messages: List[Dict], + tools: Optional[List[Dict]], + stream: bool, + custom_llm_provider: str, + kwargs: Dict, + ) -> Tuple[bool, Dict]: + """ + Check if WebSearch tool interception is needed for Chat Completions API. + + Similar to async_should_run_agentic_loop but for OpenAI-style chat completions. + """ + + verbose_logger.debug(f"WebSearchInterception: Chat completion hook called! provider={custom_llm_provider}, stream={stream}") + verbose_logger.debug(f"WebSearchInterception: Response type: {type(response)}") + + # Check if provider should be intercepted + if self.enabled_providers is not None and custom_llm_provider not in self.enabled_providers: + verbose_logger.debug( + f"WebSearchInterception: Skipping provider {custom_llm_provider} (not in enabled list: {self.enabled_providers})" + ) + return False, {} + + # Check if tools include any web search tool + has_websearch_tool = any(is_web_search_tool(t) for t in (tools or [])) + if not has_websearch_tool: + verbose_logger.debug( + "WebSearchInterception: No web search tool in request" + ) + return False, {} + + # Detect WebSearch tool_calls in response (OpenAI format) + should_intercept, tool_calls = WebSearchTransformation.transform_request( + response=response, + stream=stream, + response_format="openai", + ) + + if not should_intercept: + verbose_logger.debug( + "WebSearchInterception: No WebSearch tool_calls detected in response" + ) + return False, {} + + verbose_logger.debug( + f"WebSearchInterception: Detected {len(tool_calls)} WebSearch tool call(s), executing agentic loop" + ) + + # Return tools dict with tool calls + tools_dict = { + "tool_calls": tool_calls, + "tool_type": "websearch", + "provider": custom_llm_provider, + "response_format": "openai", } return True, tools_dict @@ -303,7 +371,11 @@ class WebSearchInterceptionLogger(CustomLogger): stream: bool, kwargs: Dict, ) -> Any: - """Execute agentic loop with WebSearch execution""" + """ + Execute agentic loop with WebSearch execution for Anthropic Messages API. + + This is the legacy method for Anthropic-style responses. + """ tool_calls = tools["tool_calls"] @@ -321,6 +393,41 @@ class WebSearchInterceptionLogger(CustomLogger): kwargs=kwargs, ) + async def async_run_chat_completion_agentic_loop( + self, + tools: Dict, + model: str, + messages: List[Dict], + response: Any, + optional_params: Dict, + logging_obj: Any, + stream: bool, + kwargs: Dict, + ) -> Any: + """ + Execute agentic loop with WebSearch execution for Chat Completions API. + + Similar to async_run_agentic_loop but for OpenAI-style chat completions. + """ + + tool_calls = tools["tool_calls"] + response_format = tools.get("response_format", "openai") + + verbose_logger.debug( + f"WebSearchInterception: Executing chat completion agentic loop for {len(tool_calls)} search(es)" + ) + + return await self._execute_chat_completion_agentic_loop( + model=model, + messages=messages, + tool_calls=tool_calls, + optional_params=optional_params, + logging_obj=logging_obj, + stream=stream, + kwargs=kwargs, + response_format=response_format, + ) + async def _execute_agentic_loop( self, model: str, @@ -521,6 +628,150 @@ class WebSearchInterceptionLogger(CustomLogger): ) raise + async def _execute_chat_completion_agentic_loop( + self, + model: str, + messages: List[Dict], + tool_calls: List[Dict], + optional_params: Dict, + logging_obj: Any, + stream: bool, + kwargs: Dict, + response_format: str = "openai", + ) -> Any: + """Execute litellm.search() and make follow-up chat completion request""" + + # Extract search queries from tool_calls + search_tasks = [] + for tool_call in tool_calls: + # Handle both Anthropic-style input and OpenAI-style function.arguments + query = None + if "input" in tool_call and isinstance(tool_call["input"], dict): + query = tool_call["input"].get("query") + elif "function" in tool_call: + func = tool_call["function"] + if isinstance(func, dict): + args = func.get("arguments", {}) + if isinstance(args, dict): + query = args.get("query") + + if query: + verbose_logger.debug( + f"WebSearchInterception: Queuing search for query='{query}'" + ) + search_tasks.append(self._execute_search(query)) + else: + verbose_logger.warning( + f"WebSearchInterception: Tool call {tool_call.get('id')} has no query" + ) + # Add empty result for tools without query + search_tasks.append(self._create_empty_search_result()) + + # Execute searches in parallel + verbose_logger.debug( + f"WebSearchInterception: Executing {len(search_tasks)} search(es) in parallel" + ) + search_results = await asyncio.gather(*search_tasks, return_exceptions=True) + + # Handle any exceptions in search results + final_search_results: List[str] = [] + for i, result in enumerate(search_results): + if isinstance(result, Exception): + verbose_logger.error( + f"WebSearchInterception: Search {i} failed with error: {str(result)}" + ) + final_search_results.append( + f"Search failed: {str(result)}" + ) + elif isinstance(result, str): + final_search_results.append(cast(str, result)) + else: + verbose_logger.warning( + f"WebSearchInterception: Unexpected result type {type(result)} at index {i}" + ) + final_search_results.append(str(result)) + + # Build assistant and tool messages using transformation + assistant_message, tool_messages_or_user = WebSearchTransformation.transform_response( + tool_calls=tool_calls, + search_results=final_search_results, + response_format=response_format, + ) + + # Make follow-up request with search results + # For OpenAI format, tool_messages_or_user is a list of tool messages + if response_format == "openai": + follow_up_messages = messages + [assistant_message] + tool_messages_or_user + else: + # For Anthropic format (shouldn't happen in this method, but handle it) + follow_up_messages = messages + [assistant_message, tool_messages_or_user] + + verbose_logger.debug( + "WebSearchInterception: Making follow-up chat completion request with search results" + ) + verbose_logger.debug( + f"WebSearchInterception: Follow-up messages count: {len(follow_up_messages)}" + ) + + # Use litellm.acompletion for follow-up request + try: + # Remove internal parameters that shouldn't be passed to follow-up request + internal_params = { + '_websearch_interception', + 'acompletion', + 'litellm_logging_obj', + 'custom_llm_provider', + 'model_alias_map', + 'stream_response', + 'custom_prompt_dict', + } + kwargs_for_followup = { + k: v for k, v in kwargs.items() + if not k.startswith('_websearch_interception') and k not in internal_params + } + + # Get full model name from kwargs + full_model_name = model + if "custom_llm_provider" in kwargs: + custom_llm_provider = kwargs["custom_llm_provider"] + # Reconstruct full model name with provider prefix if needed + if not model.startswith(custom_llm_provider): + # Check if model already has a provider prefix + if "/" not in model: + full_model_name = f"{custom_llm_provider}/{model}" + + verbose_logger.debug( + f"WebSearchInterception: Using model name: {full_model_name}" + ) + + # Prepare tools for follow-up request (same as original) + tools_param = optional_params.get("tools") + + # Remove tools and extra_body from optional_params to avoid issues + # extra_body often contains internal LiteLLM params that shouldn't be forwarded + optional_params_clean = { + k: v for k, v in optional_params.items() + if k not in {"tools", "extra_body", "model_alias_map","stream_response", "custom_prompt_dict" } + } + + final_response = await litellm.acompletion( + model=full_model_name, + messages=follow_up_messages, + tools=tools_param, + **optional_params_clean, + **kwargs_for_followup, + ) + + verbose_logger.debug( + f"WebSearchInterception: Follow-up request completed, response type: {type(final_response)}" + ) + return final_response + except Exception as e: + verbose_logger.exception( + f"WebSearchInterception: Follow-up request failed: {str(e)}" + ) + raise + async def _create_empty_search_result(self) -> str: """Create an empty search result for tool calls without queries""" return "No search query provided" From ea4e48e13a6d7b6a21d087e836201a52c55f2d72 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:28:06 +0530 Subject: [PATCH 02/50] Add chat completion tool calls support and response transformation --- .../websearch_interception/transformation.py | 185 ++++++++++++++++-- 1 file changed, 171 insertions(+), 14 deletions(-) diff --git a/litellm/integrations/websearch_interception/transformation.py b/litellm/integrations/websearch_interception/transformation.py index 313358822a5..0884d408c84 100644 --- a/litellm/integrations/websearch_interception/transformation.py +++ b/litellm/integrations/websearch_interception/transformation.py @@ -1,7 +1,7 @@ """ WebSearch Tool Transformation -Transforms between Anthropic tool_use format and LiteLLM search format. +Transforms between Anthropic/OpenAI tool_use format and LiteLLM search format. """ from typing import Any, Dict, List, Tuple @@ -17,28 +17,31 @@ class WebSearchTransformation: Handles transformation between: - Anthropic tool_use format β†’ LiteLLM search requests - - LiteLLM SearchResponse β†’ Anthropic tool_result format + - OpenAI tool_calls format β†’ LiteLLM search requests + - LiteLLM SearchResponse β†’ Anthropic/OpenAI tool_result format """ @staticmethod def transform_request( response: Any, stream: bool, + response_format: str = "anthropic", ) -> Tuple[bool, List[Dict]]: """ - Transform Anthropic response to extract WebSearch tool calls. + Transform model response to extract WebSearch tool calls. - Detects if response contains WebSearch tool_use blocks and extracts + Detects if response contains WebSearch tool_use/tool_calls blocks and extracts the search queries for execution. Args: - response: Model response (dict or AnthropicMessagesResponse) + response: Model response (dict, AnthropicMessagesResponse, or ModelResponse) stream: Whether response is streaming + response_format: Response format - "anthropic" or "openai" (default: "anthropic") Returns: (has_websearch, tool_calls): has_websearch: True if WebSearch tool_use found - tool_calls: List of tool_use dicts with id, name, input + tool_calls: List of tool_use/tool_calls dicts with id, name, input/function Note: Streaming requests are handled by converting stream=True to stream=False @@ -54,8 +57,11 @@ class WebSearchTransformation: ) return False, [] - # Parse non-streaming response - return WebSearchTransformation._detect_from_non_streaming_response(response) + # Parse non-streaming response based on format + if response_format == "openai": + return WebSearchTransformation._detect_from_openai_response(response) + else: + return WebSearchTransformation._detect_from_non_streaming_response(response) @staticmethod def _detect_from_non_streaming_response( @@ -114,26 +120,143 @@ class WebSearchTransformation: return len(tool_calls) > 0, tool_calls + @staticmethod + def _detect_from_openai_response( + response: Any, + ) -> Tuple[bool, List[Dict]]: + """Parse OpenAI-style response for WebSearch tool_calls""" + + # Handle both dict and ModelResponse objects + if isinstance(response, dict): + choices = response.get("choices", []) + else: + if not hasattr(response, "choices"): + verbose_logger.debug( + "WebSearchInterception: Response has no choices attribute" + ) + return False, [] + choices = response.choices or [] + + if not choices: + verbose_logger.debug( + "WebSearchInterception: Response has empty choices" + ) + return False, [] + + # Get first choice's message + first_choice = choices[0] + if isinstance(first_choice, dict): + message = first_choice.get("message", {}) + else: + message = getattr(first_choice, "message", None) + + if not message: + verbose_logger.debug( + "WebSearchInterception: First choice has no message" + ) + return False, [] + + # Get tool_calls from message + if isinstance(message, dict): + openai_tool_calls = message.get("tool_calls", []) + else: + openai_tool_calls = getattr(message, "tool_calls", None) or [] + + if not openai_tool_calls: + verbose_logger.debug( + "WebSearchInterception: Message has no tool_calls" + ) + return False, [] + + # Find all WebSearch tool calls + tool_calls = [] + for tool_call in openai_tool_calls: + # Handle both dict and object tool calls + if isinstance(tool_call, dict): + tool_id = tool_call.get("id") + tool_type = tool_call.get("type") + function = tool_call.get("function", {}) + function_name = function.get("name") if isinstance(function, dict) else getattr(function, "name", None) + function_arguments = function.get("arguments") if isinstance(function, dict) else getattr(function, "arguments", None) + else: + tool_id = getattr(tool_call, "id", None) + tool_type = getattr(tool_call, "type", None) + function = getattr(tool_call, "function", None) + function_name = getattr(function, "name", None) if function else None + function_arguments = getattr(function, "arguments", None) if function else None + + # Check for LiteLLM standard or legacy web search tools + if tool_type == "function" and function_name in ( + LITELLM_WEB_SEARCH_TOOL_NAME, "WebSearch", "web_search" + ): + # Parse arguments (might be JSON string) + import json + if isinstance(function_arguments, str): + try: + arguments = json.loads(function_arguments) + except json.JSONDecodeError: + verbose_logger.warning( + f"WebSearchInterception: Failed to parse function arguments: {function_arguments}" + ) + arguments = {} + else: + arguments = function_arguments or {} + + # Convert to internal format (similar to Anthropic) + tool_call_dict = { + "id": tool_id, + "type": "function", + "name": function_name, + "function": { + "name": function_name, + "arguments": arguments, + }, + "input": arguments, # For compatibility with Anthropic format + } + tool_calls.append(tool_call_dict) + verbose_logger.debug( + f"WebSearchInterception: Found {function_name} tool_call with id={tool_id}" + ) + + return len(tool_calls) > 0, tool_calls + @staticmethod def transform_response( tool_calls: List[Dict], search_results: List[str], + response_format: str = "anthropic", ) -> Tuple[Dict, Dict]: """ - Transform LiteLLM search results to Anthropic tool_result format. + Transform LiteLLM search results to Anthropic/OpenAI tool_result format. - Builds the assistant and user messages needed for the agentic loop + Builds the assistant and user/tool messages needed for the agentic loop follow-up request. Args: - tool_calls: List of tool_use dicts from transform_request + tool_calls: List of tool_use/tool_calls dicts from transform_request search_results: List of search result strings (one per tool_call) + response_format: Response format - "anthropic" or "openai" (default: "anthropic") Returns: - (assistant_message, user_message): - assistant_message: Message with tool_use blocks - user_message: Message with tool_result blocks + (assistant_message, user_or_tool_messages): + For Anthropic: assistant_message with tool_use blocks, user_message with tool_result blocks + For OpenAI: assistant_message with tool_calls, tool_messages list with tool results """ + if response_format == "openai": + return WebSearchTransformation._transform_response_openai( + tool_calls, search_results + ) + else: + return WebSearchTransformation._transform_response_anthropic( + tool_calls, search_results + ) + + @staticmethod + def _transform_response_anthropic( + tool_calls: List[Dict], + search_results: List[str], + ) -> Tuple[Dict, Dict]: + """Transform to Anthropic format (single user message with tool_result blocks)""" # Build assistant message with tool_use blocks assistant_message = { "role": "assistant", @@ -163,6 +286,40 @@ class WebSearchTransformation: return assistant_message, user_message + @staticmethod + def _transform_response_openai( + tool_calls: List[Dict], + search_results: List[str], + ) -> Tuple[Dict, List[Dict]]: + """Transform to OpenAI format (assistant with tool_calls, separate tool messages)""" + # Build assistant message with tool_calls + assistant_message = { + "role": "assistant", + "tool_calls": [ + { + "id": tc["id"], + "type": "function", + "function": { + "name": tc["name"], + "arguments": str(tc["input"]), + }, + } + for tc in tool_calls + ], + } + + # Build separate tool messages (one per tool call) + tool_messages = [ + { + "role": "tool", + "tool_call_id": tool_calls[i]["id"], + "content": search_results[i], + } + for i in range(len(tool_calls)) + ] + + return assistant_message, tool_messages + @staticmethod def format_search_response(result: SearchResponse) -> str: """ From 245d705e6ca7ee9c8cda2dccb2d40256480726f2 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:28:28 +0530 Subject: [PATCH 03/50] Add new methods in chat completion --- litellm/integrations/custom_logger.py | 31 +++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index 07d237c4758..4a341863d4b 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -664,6 +664,37 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac return final_response """ pass + + async def async_should_run_chat_completion_agentic_loop( + self, + response: Any, + model: str, + messages: List[Dict], + tools: Optional[List[Dict]], + stream: bool, + custom_llm_provider: str, + kwargs: Dict, + ) -> Tuple[bool, Dict]: + """ + Hook to determine if chat completion agentic loop should be executed. + """ + return False, {} + + async def async_run_chat_completion_agentic_loop( + self, + tools: Dict, + model: str, + messages: List[Dict], + response: Any, + optional_params: Dict, + logging_obj: "LiteLLMLoggingObj", + stream: bool, + kwargs: Dict, + ) -> Any: + """ + Hook to execute chat completion agentic loop based on context from should_run hook. + """ + pass # Useful helpers for custom logger classes From 6207bf8f6856c185844c481aebe9f5eebb261801 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:28:43 +0530 Subject: [PATCH 04/50] Add chat completion tool format --- .../integrations/websearch_interception/tools.py | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/litellm/integrations/websearch_interception/tools.py b/litellm/integrations/websearch_interception/tools.py index 4f8b7372fe3..c92c66f41ee 100644 --- a/litellm/integrations/websearch_interception/tools.py +++ b/litellm/integrations/websearch_interception/tools.py @@ -55,6 +55,7 @@ def is_web_search_tool(tool: Dict[str, Any]) -> bool: Detects: - LiteLLM standard: name == "litellm_web_search" + - OpenAI format: type == "function" with function.name == "litellm_web_search" - Anthropic native: type starts with "web_search_" (e.g., "web_search_20250305") - Claude Code: name == "web_search" with a type field - Custom: name == "WebSearch" (legacy format) @@ -68,15 +69,25 @@ def is_web_search_tool(tool: Dict[str, Any]) -> bool: Example: >>> is_web_search_tool({"name": "litellm_web_search"}) True + >>> is_web_search_tool({"type": "function", "function": {"name": "litellm_web_search"}}) + True >>> is_web_search_tool({"type": "web_search_20250305", "name": "web_search"}) True >>> is_web_search_tool({"name": "calculator"}) False """ + print(f"πŸ”₯tool: {tool}") tool_name = tool.get("name", "") tool_type = tool.get("type", "") + + # Check for OpenAI format: {"type": "function", "function": {"name": "..."}} + if tool_type == "function" and "function" in tool: + function_def = tool.get("function", {}) + function_name = function_def.get("name", "") + if function_name == LITELLM_WEB_SEARCH_TOOL_NAME: + return True - # Check for LiteLLM standard tool + # Check for LiteLLM standard tool (Anthropic format) if tool_name == LITELLM_WEB_SEARCH_TOOL_NAME: return True From 88778a871dce4378f84e56fbae5b10e60476dd5e Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:29:21 +0530 Subject: [PATCH 05/50] Add callback for websearch in completion method --- litellm/llms/custom_httpx/llm_http_handler.py | 128 +++++++++++++++++- litellm/llms/openai/openai.py | 92 ++++++++++++- 2 files changed, 214 insertions(+), 6 deletions(-) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index d2ea7e872a2..3907ff7abf7 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -302,7 +302,7 @@ class BaseLLMHTTPHandler: logging_obj=logging_obj, signed_json_body=signed_json_body, ) - return provider_config.transform_response( + initial_response = provider_config.transform_response( model=model, raw_response=response, model_response=model_response, @@ -316,6 +316,20 @@ class BaseLLMHTTPHandler: json_mode=json_mode, ) + # Call agentic chat completion hooks + final_response = await self._call_agentic_chat_completion_hooks( + response=initial_response, + model=model, + messages=messages, + optional_params=optional_params, + logging_obj=logging_obj, + stream=False, + custom_llm_provider=custom_llm_provider, + kwargs=litellm_params, + ) + + return final_response if final_response is not None else initial_response + def completion( self, model: str, @@ -412,6 +426,11 @@ class BaseLLMHTTPHandler: }, ) + # Check if stream was converted for WebSearch interception + # This is set by the async_pre_request_hook in WebSearchInterceptionLogger + if litellm_params.get("_websearch_interception_converted_stream", False): + logging_obj.model_call_details["websearch_interception_converted_stream"] = True + if acompletion is True: if stream is True: data = self._add_stream_param_to_request_body( @@ -419,7 +438,7 @@ class BaseLLMHTTPHandler: provider_config=provider_config, fake_stream=fake_stream, ) - return self.acompletion_stream_function( + response = self.acompletion_stream_function( model=model, messages=messages, api_base=api_base, @@ -4361,10 +4380,10 @@ class BaseLLMHTTPHandler: kwargs: Dict, ) -> Optional[Any]: """ - Call agentic completion hooks for all custom loggers. + Call agentic completion hooks for all custom loggers (Anthropic Messages API). - 1. Call async_should_run_agentic_completion to check if agentic loop is needed - 2. If yes, call async_run_agentic_completion to execute the loop + 1. Call async_should_run_agentic_loop to check if agentic loop is needed + 2. If yes, call async_run_agentic_loop to execute the loop Returns the response from agentic loop, or None if no hook runs. """ @@ -4453,6 +4472,105 @@ class BaseLLMHTTPHandler: return None + async def _call_agentic_chat_completion_hooks( + self, + response: Any, + model: str, + messages: List[Dict], + optional_params: Dict, + logging_obj: "LiteLLMLoggingObj", + stream: bool, + custom_llm_provider: str, + kwargs: Dict, + ) -> Optional[Any]: + """ + Call agentic chat completion hooks for all custom loggers (Chat Completions API). + + 1. Call async_should_run_chat_completion_agentic_loop to check if agentic loop is needed + 2. If yes, call async_run_chat_completion_agentic_loop to execute the loop + + Returns the response from agentic loop, or None if no hook runs. + """ + from litellm._logging import verbose_logger + from litellm.integrations.custom_logger import CustomLogger + + callbacks = litellm.callbacks + ( + logging_obj.dynamic_success_callbacks or [] + ) + tools = optional_params.get("tools", []) + + for callback in callbacks: + try: + if isinstance(callback, CustomLogger): + # Check if callback has the chat completion agentic loop method + if not hasattr(callback, "async_should_run_chat_completion_agentic_loop"): + continue + + # First: Check if agentic loop should run + should_run, tool_calls = ( + await callback.async_should_run_chat_completion_agentic_loop( + response=response, + model=model, + messages=messages, + tools=tools, + stream=stream, + custom_llm_provider=custom_llm_provider, + kwargs=kwargs, + ) + ) + + if should_run: + # Second: Execute agentic loop + # Add custom_llm_provider to kwargs so the agentic loop can reconstruct the full model name + kwargs_with_provider = kwargs.copy() if kwargs else {} + kwargs_with_provider["custom_llm_provider"] = custom_llm_provider + agentic_response = await callback.async_run_chat_completion_agentic_loop( + tools=tool_calls, + model=model, + messages=messages, + response=response, + optional_params=optional_params, + logging_obj=logging_obj, + stream=stream, + kwargs=kwargs_with_provider, + ) + # First hook that runs agentic loop wins + return agentic_response + + except Exception as e: + verbose_logger.exception( + f"LiteLLM.AgenticHookError: Exception in chat completion agentic hooks: {str(e)}" + ) + + # Check if we need to convert response to fake stream for chat completions + # This happens when: + # 1. Stream was originally True but converted to False for WebSearch interception + # 2. No agentic loop ran (LLM didn't use the tool) + # 3. We have a non-streaming response that needs to be converted to streaming + websearch_converted_stream = ( + logging_obj.model_call_details.get("websearch_interception_converted_stream", False) + if logging_obj is not None + else False + ) + + if websearch_converted_stream: + from litellm._logging import verbose_logger + from litellm.llms.base_llm.base_model_iterator import ( + convert_model_response_to_streaming, + ) + + verbose_logger.debug( + "WebSearchInterception: No tool call made, converting non-streaming chat completion to fake stream" + ) + + # Convert the non-streaming ModelResponse to a fake stream + if hasattr(response, "choices"): + # Use the existing converter for ModelResponse + fake_stream = convert_model_response_to_streaming(response) + return fake_stream + + return None + def _handle_error( self, e: Exception, diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 8a8070240da..2f0e5e480b5 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -501,6 +501,82 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): else: raise e + async def _call_agentic_completion_hooks_openai( + self, + response: Any, + model: str, + messages: List[Dict], + optional_params: Dict, + logging_obj: LiteLLMLoggingObj, + stream: bool, + litellm_params: Dict, + ) -> Optional[Any]: + """ + Call agentic completion hooks for all custom loggers (OpenAI Chat Completions API). + + 1. Call async_should_run_chat_completion_agentic_loop to check if agentic loop is needed + 2. If yes, call async_run_chat_completion_agentic_loop to execute the loop + + Returns the response from agentic loop, or None if no hook runs. + """ + from litellm._logging import verbose_logger + from litellm.integrations.custom_logger import CustomLogger + + callbacks = litellm.callbacks + ( + logging_obj.dynamic_success_callbacks or [] + ) + print(f"πŸ”₯callbacks: {callbacks}") + tools = optional_params.get("tools", []) + print(f"πŸ”₯tools: {tools}") + # Get custom_llm_provider from litellm_params + custom_llm_provider = litellm_params.get("custom_llm_provider", "openai") + + for callback in callbacks: + try: + if isinstance(callback, CustomLogger): + # Check if the callback has the chat completion agentic loop methods + if not hasattr(callback, 'async_should_run_chat_completion_agentic_loop'): + continue + + # First: Check if agentic loop should run (using chat completion method) + should_run, tool_calls = ( + await callback.async_should_run_chat_completion_agentic_loop( + response=response, + model=model, + messages=messages, + tools=tools, + stream=stream, + custom_llm_provider=custom_llm_provider, + kwargs=litellm_params, + ) + ) + + if should_run: + # Second: Execute agentic loop + kwargs_with_provider = litellm_params.copy() if litellm_params else {} + kwargs_with_provider["custom_llm_provider"] = custom_llm_provider + + # For OpenAI Chat Completions, use the chat completion agentic loop method + agentic_response = await callback.async_run_chat_completion_agentic_loop( + tools=tool_calls, + model=model, + messages=messages, + response=response, + optional_params=optional_params, + logging_obj=logging_obj, + stream=stream, + kwargs=kwargs_with_provider, + ) + # First hook that runs agentic loop wins + return agentic_response + + except Exception as e: + verbose_logger.exception( + f"LiteLLM.AgenticHookError: Exception in agentic completion hooks for OpenAI: {str(e)}" + ) + + return None + def mock_streaming( self, response: ModelResponse, @@ -844,7 +920,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): logging_obj=logging_obj, ) stringified_response = response.model_dump() - + print(f"πŸ”₯stringified_response: {stringified_response}") logging_obj.post_call( input=data["messages"], api_key=api_key, @@ -859,6 +935,20 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): _response_headers=headers, ) + # Call agentic completion hooks (e.g., for websearch_interception) + agentic_response = await self._call_agentic_completion_hooks_openai( + response=final_response_obj, + model=model, + messages=messages, + optional_params=optional_params, + logging_obj=logging_obj, + stream=False, + litellm_params=litellm_params, + ) + + if agentic_response is not None: + final_response_obj = agentic_response + if fake_stream is True: return self.mock_streaming( response=cast(ModelResponse, final_response_obj), From 4b0eb50ddddfe3bc0b70c4ef649391be603e8923 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 5 Feb 2026 14:29:31 +0530 Subject: [PATCH 06/50] Add test for web search --- test_websearch_chat_completion.py | 136 ++++++++++++++++++++++++++++++ 1 file changed, 136 insertions(+) create mode 100644 test_websearch_chat_completion.py diff --git a/test_websearch_chat_completion.py b/test_websearch_chat_completion.py new file mode 100644 index 00000000000..e572e4d860c --- /dev/null +++ b/test_websearch_chat_completion.py @@ -0,0 +1,136 @@ +""" +Test script for WebSearch interception with chat completions API. + +This script demonstrates how to use the websearch_interception callback +with litellm.acompletion() for transparent server-side web search execution. +""" +import asyncio +import litellm + +# Enable verbose logging to see what's happening +litellm.set_verbose = True + + +async def test_websearch_chat_completion(): + """Test websearch interception with chat completions API.""" + + # Configure WebSearch interception + litellm.callbacks = ["websearch_interception"] + + print("\n" + "="*80) + print("Testing WebSearch Interception with Chat Completions API") + print("="*80 + "\n") + + # User makes a simple completion call with tools + print("Making request to GPT-4o with litellm_web_search tool...") + print("Question: What's the weather in San Francisco today?") + print("\nExpected behavior:") + print("1. Model calls litellm_web_search tool") + print("2. Server executes web search automatically") + print("3. Server makes follow-up request with search results") + print("4. User gets final answer\n") + + response = await litellm.acompletion( + model="gpt-4o", + messages=[ + {"role": "user", "content": "What's the weather in San Francisco today?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web for information", + "parameters": { + "type": "object", + "properties": { + "query": {"type": "string", "description": "Search query"} + }, + "required": ["query"] + } + } + } + ] + ) + + print("\n" + "-"*80) + print("FINAL RESPONSE:") + print("-"*80) + print(f"\nContent: {response.choices[0].message.content}") + print(f"\nFinish reason: {response.choices[0].finish_reason}") + + # Check if we got tool_calls (should NOT if agentic loop worked) + if hasattr(response.choices[0].message, 'tool_calls') and response.choices[0].message.tool_calls: + print("\n⚠️ WARNING: Got tool_calls in response!") + print("This means the agentic loop did NOT execute automatically.") + print(f"Tool calls: {response.choices[0].message.tool_calls}") + else: + print("\nβœ… SUCCESS: No tool_calls in response!") + print("The agentic loop executed automatically and returned the final answer.") + + print("\n" + "="*80 + "\n") + + +async def test_streaming_websearch(): + """Test websearch interception with streaming.""" + + # Configure WebSearch interception + litellm.callbacks = ["websearch_interception"] + + print("\n" + "="*80) + print("Testing WebSearch Interception with STREAMING") + print("="*80 + "\n") + + print("Making STREAMING request to GPT-4o with litellm_web_search tool...") + print("Question: What are the latest AI news?") + + response = await litellm.acompletion( + model="gpt-4o", + messages=[ + {"role": "user", "content": "What are the latest AI news from today?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web for information", + "parameters": { + "type": "object", + "properties": { + "query": {"type": "string"} + } + } + } + } + ], + stream=True + ) + + print("\n" + "-"*80) + print("STREAMING RESPONSE:") + print("-"*80 + "\n") + + full_content = "" + async for chunk in response: + if hasattr(chunk.choices[0].delta, 'content') and chunk.choices[0].delta.content: + content = chunk.choices[0].delta.content + print(content, end="", flush=True) + full_content += content + + print("\n\nβœ… Streaming completed successfully!") + print(f"Total content length: {len(full_content)} chars") + print("\n" + "="*80 + "\n") + + +if __name__ == "__main__": + print("\nWebSearch Interception Test Suite") + print("==================================\n") + print("This test demonstrates transparent server-side web search execution.") + print("The agentic loop happens automatically - user just gets the final answer.\n") + + # Run tests + asyncio.run(test_websearch_chat_completion()) + + # Uncomment to test streaming + # asyncio.run(test_streaming_websearch()) From a2e70a561d103497fed719829c48eb597142f214 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 6 Feb 2026 08:20:50 +0530 Subject: [PATCH 07/50] Potential fix for code scanning alert no. 4046: Clear-text logging of sensitive information Co-authored-by: Copilot Autofix powered by AI <62310815+github-advanced-security[bot]@users.noreply.github.com> --- litellm/llms/openai/openai.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 2f0e5e480b5..c6f502d3a25 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -525,9 +525,15 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): callbacks = litellm.callbacks + ( logging_obj.dynamic_success_callbacks or [] ) - print(f"πŸ”₯callbacks: {callbacks}") + # Avoid logging full callback objects to prevent leaking sensitive data + verbose_logger.debug( + "LiteLLM.AgenticHooks: callbacks_count=%s", len(callbacks) + ) tools = optional_params.get("tools", []) - print(f"πŸ”₯tools: {tools}") + # Avoid logging full tools payloads; they may contain sensitive parameters + verbose_logger.debug( + "LiteLLM.AgenticHooks: tools_count=%s", len(tools) if isinstance(tools, list) else 1 if tools else 0 + ) # Get custom_llm_provider from litellm_params custom_llm_provider = litellm_params.get("custom_llm_provider", "openai") From f12875bd428d74cb46cc24d19ec44d4f3bdafc32 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 6 Feb 2026 08:21:09 +0530 Subject: [PATCH 08/50] Update litellm/integrations/websearch_interception/tools.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/integrations/websearch_interception/tools.py | 1 - 1 file changed, 1 deletion(-) diff --git a/litellm/integrations/websearch_interception/tools.py b/litellm/integrations/websearch_interception/tools.py index c92c66f41ee..be8808622da 100644 --- a/litellm/integrations/websearch_interception/tools.py +++ b/litellm/integrations/websearch_interception/tools.py @@ -76,7 +76,6 @@ def is_web_search_tool(tool: Dict[str, Any]) -> bool: >>> is_web_search_tool({"name": "calculator"}) False """ - print(f"πŸ”₯tool: {tool}") tool_name = tool.get("name", "") tool_type = tool.get("type", "") From 51d565f619604cb60d2b50f6f19721da7d35f41c Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Sat, 7 Feb 2026 03:10:53 +0530 Subject: [PATCH 09/50] fix conflicts with main- (this PR is from upstream/main) --- litellm/proxy/_types.py | 16 +- .../proxy/hooks/model_max_budget_limiter.py | 38 ++- .../budget_management_endpoints.py | 39 ++- .../key_management_endpoints.py | 233 ++++++++++-------- tests/proxy_unit_tests/test_proxy_utils.py | 145 +++++++---- ...test_unit_test_max_model_budget_limiter.py | 63 +++-- .../test_budget_endpoints.py | 54 ++-- 7 files changed, 393 insertions(+), 195 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index f38f94f4c98..7324c3ea1de 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -359,7 +359,6 @@ class LiteLLMRoutes(enum.Enum): "/v1/vector_stores/{vector_store_id}/files/{file_id}/content", "/vector_store/list", "/v1/vector_store/list", - # search "/search", "/v1/search", @@ -2220,13 +2219,22 @@ class LiteLLM_VerificationTokenView(LiteLLM_VerificationToken): last_refreshed_at: Optional[float] = None # last time joint view was pulled from db def __init__(self, **kwargs): - # Handle litellm_budget_table_* keys + # Handle litellm_budget_table_* keys (budget table overrides when key value is None or empty) for key, value in list(kwargs.items()): if key.startswith("litellm_budget_table_") and value is not None: # Extract the corresponding attribute name attr_name = key.replace("litellm_budget_table_", "") - # Check if the value is None and set the corresponding attribute - if getattr(self, attr_name, None) is None: + # Use key's value from kwargs (from DB view), not class default + current = kwargs.get(attr_name) + if current is None: + current = getattr(self, attr_name, None) + # Apply budget value when key has no value, or for model_max_budget when key has empty dict + should_apply = current is None or ( + attr_name == "model_max_budget" + and isinstance(current, dict) + and len(current) == 0 + ) + if should_apply: kwargs[attr_name] = value if key == "end_user_id" and value is not None and isinstance(value, int): kwargs[key] = str(value) diff --git a/litellm/proxy/hooks/model_max_budget_limiter.py b/litellm/proxy/hooks/model_max_budget_limiter.py index ac02c915366..69c7e92d82e 100644 --- a/litellm/proxy/hooks/model_max_budget_limiter.py +++ b/litellm/proxy/hooks/model_max_budget_limiter.py @@ -171,19 +171,35 @@ class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting): return response_cost: float = standard_logging_payload.get("response_cost", 0) model = standard_logging_payload.get("model") + virtual_key = standard_logging_payload.get("metadata", {}).get( + "user_api_key_hash" + ) - virtual_key = standard_logging_payload.get("metadata").get("user_api_key_hash") - model = standard_logging_payload.get("model") - if virtual_key is not None: - budget_config = BudgetConfig(time_period="1d", budget_limit=0.1) - virtual_spend_key = f"{VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX}:{virtual_key}:{model}:{budget_config.budget_duration}" - virtual_start_time_key = f"virtual_key_budget_start_time:{virtual_key}" - await self._increment_spend_for_key( - budget_config=budget_config, - spend_key=virtual_spend_key, - start_time_key=virtual_start_time_key, - response_cost=response_cost, + if virtual_key is None or model is None: + return + + # Resolve per-model budget config (same logic as is_key_within_model_budget) + internal_model_max_budget: GenericBudgetConfigType = {} + for _model, _budget_info in user_api_key_model_max_budget.items(): + internal_model_max_budget[_model] = BudgetConfig(**_budget_info) + key_budget_config = self._get_request_model_budget_config( + model=model, internal_model_max_budget=internal_model_max_budget + ) + if key_budget_config is None or not key_budget_config.budget_duration: + verbose_proxy_logger.debug( + "Not incrementing model spend: no budget config or budget_duration for model=%s", + model, ) + return + + virtual_spend_key = f"{VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX}:{virtual_key}:{model}:{key_budget_config.budget_duration}" + virtual_start_time_key = f"virtual_key_budget_start_time:{virtual_key}" + await self._increment_spend_for_key( + budget_config=key_budget_config, + spend_key=virtual_spend_key, + start_time_key=virtual_start_time_key, + response_cost=response_cost, + ) verbose_proxy_logger.debug( "current state of in memory cache %s", json.dumps( diff --git a/litellm/proxy/management_endpoints/budget_management_endpoints.py b/litellm/proxy/management_endpoints/budget_management_endpoints.py index e43da32565a..20c7f9ec412 100644 --- a/litellm/proxy/management_endpoints/budget_management_endpoints.py +++ b/litellm/proxy/management_endpoints/budget_management_endpoints.py @@ -59,14 +59,29 @@ async def new_budget( if budget_obj.max_budget is not None and budget_obj.max_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"max_budget cannot be negative. Received: {budget_obj.max_budget}"} + detail={ + "error": f"max_budget cannot be negative. Received: {budget_obj.max_budget}" + }, ) if budget_obj.soft_budget is not None and budget_obj.soft_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"soft_budget cannot be negative. Received: {budget_obj.soft_budget}"} + detail={ + "error": f"soft_budget cannot be negative. Received: {budget_obj.soft_budget}" + }, ) + # Validate model_max_budget if present + if budget_obj.model_max_budget is not None and len(budget_obj.model_max_budget) > 0: + from litellm.proxy.management_endpoints.key_management_endpoints import ( + validate_model_max_budget, + ) + + try: + validate_model_max_budget(budget_obj.model_max_budget) + except ValueError as e: + raise HTTPException(status_code=400, detail={"error": str(e)}) + # if no budget_reset_at date is set, but a budget_duration is given, then set budget_reset_at initially to the first completed duration interval in future if budget_obj.budget_reset_at is None and budget_obj.budget_duration is not None: budget_obj.budget_reset_at = datetime.utcnow() + timedelta( @@ -123,14 +138,29 @@ async def update_budget( if budget_obj.max_budget is not None and budget_obj.max_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"max_budget cannot be negative. Received: {budget_obj.max_budget}"} + detail={ + "error": f"max_budget cannot be negative. Received: {budget_obj.max_budget}" + }, ) if budget_obj.soft_budget is not None and budget_obj.soft_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"soft_budget cannot be negative. Received: {budget_obj.soft_budget}"} + detail={ + "error": f"soft_budget cannot be negative. Received: {budget_obj.soft_budget}" + }, ) + # Validate model_max_budget if present in update + if budget_obj.model_max_budget is not None and len(budget_obj.model_max_budget) > 0: + from litellm.proxy.management_endpoints.key_management_endpoints import ( + validate_model_max_budget, + ) + + try: + validate_model_max_budget(budget_obj.model_max_budget) + except ValueError as e: + raise HTTPException(status_code=400, detail={"error": str(e)}) + response = await prisma_client.db.litellm_budgettable.update( where={"budget_id": budget_obj.budget_id}, data={ @@ -226,6 +256,7 @@ async def budget_settings( "budget_duration": {"type": "String"}, "max_budget": {"type": "Float"}, "soft_budget": {"type": "Float"}, + "model_max_budget": {"type": "Object"}, } return_val = [] diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 9dadffca351..f1c97693455 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -518,7 +518,7 @@ async def _common_key_generation_helper( # noqa: PLR0915 ) # Handle special case where duration is "-1" (never expires) if value == "-1": - user_duration = float('inf') # Infinite duration + user_duration = float("inf") # Infinite duration else: user_duration = duration_in_seconds(duration=value) if user_duration > upperbound_duration: @@ -660,9 +660,9 @@ async def _common_key_generation_helper( # noqa: PLR0915 request_type="key", **data_json, table_name="key" ) - response["soft_budget"] = ( - data.soft_budget - ) # include the user-input soft budget in the response + response[ + "soft_budget" + ] = data.soft_budget # include the user-input soft budget in the response response = GenerateKeyResponse(**response) @@ -1083,12 +1083,16 @@ async def generate_key_fn( if data.max_budget is not None and data.max_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"max_budget cannot be negative. Received: {data.max_budget}"} + detail={ + "error": f"max_budget cannot be negative. Received: {data.max_budget}" + }, ) if data.soft_budget is not None and data.soft_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"soft_budget cannot be negative. Received: {data.soft_budget}"} + detail={ + "error": f"soft_budget cannot be negative. Received: {data.soft_budget}" + }, ) if user_custom_key_generate is not None: @@ -1399,8 +1403,13 @@ async def prepare_key_update_data( validate_model_max_budget(non_default_values["model_max_budget"]) # Serialize router_settings to JSON if present - if "router_settings" in non_default_values and non_default_values["router_settings"] is not None: - non_default_values["router_settings"] = safe_dumps(non_default_values["router_settings"]) + if ( + "router_settings" in non_default_values + and non_default_values["router_settings"] is not None + ): + non_default_values["router_settings"] = safe_dumps( + non_default_values["router_settings"] + ) non_default_values = prepare_metadata_fields( data=data, non_default_values=non_default_values, existing_metadata=_metadata @@ -1448,19 +1457,17 @@ def is_different_team( def _validate_max_budget(max_budget: Optional[float]) -> None: """ Validate that max_budget is not negative. - + Args: max_budget: The max_budget value to validate - + Raises: HTTPException: If max_budget is negative """ if max_budget is not None and max_budget < 0: raise HTTPException( status_code=400, - detail={ - "error": f"max_budget cannot be negative. Received: {max_budget}" - }, + detail={"error": f"max_budget cannot be negative. Received: {max_budget}"}, ) @@ -1469,14 +1476,14 @@ async def _get_and_validate_existing_key( ) -> LiteLLM_VerificationToken: """ Get existing key from database and validate it exists. - + Args: token: The key token to look up prisma_client: Prisma client instance - + Returns: LiteLLM_VerificationToken: The existing key row - + Raises: HTTPException: If key is not found """ @@ -1485,19 +1492,19 @@ async def _get_and_validate_existing_key( status_code=500, detail={"error": "Database not connected"}, ) - + existing_key_row = await prisma_client.get_data( token=token, table_name="key", query_type="find_unique", ) - + if existing_key_row is None: raise HTTPException( status_code=404, detail={"error": f"Key not found: {token}"}, ) - + return existing_key_row @@ -1512,10 +1519,10 @@ async def _process_single_key_update( ) -> Dict[str, Any]: """ Process a single key update with all validations and checks. - + This function encapsulates all the logic for updating a single key, including validation, permission checks, team checks, and database updates. - + Args: key_update_item: The key update request item user_api_key_dict: The authenticated user's API key info @@ -1524,22 +1531,22 @@ async def _process_single_key_update( user_api_key_cache: User API key cache proxy_logging_obj: Proxy logging object llm_router: LLM router instance - + Returns: Dict containing the updated key information - + Raises: HTTPException: For various validation and permission errors """ # Validate max_budget _validate_max_budget(key_update_item.max_budget) - + # Get and validate existing key existing_key_row = await _get_and_validate_existing_key( token=key_update_item.key, prisma_client=prisma_client, ) - + # Check team member permissions if prisma_client is not None: await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint( @@ -1549,7 +1556,7 @@ async def _process_single_key_update( existing_key_row=existing_key_row, user_api_key_cache=user_api_key_cache, ) - + # Create UpdateKeyRequest from BulkUpdateKeyRequestItem update_key_request = UpdateKeyRequest( key=key_update_item.key, @@ -1558,7 +1565,7 @@ async def _process_single_key_update( team_id=key_update_item.team_id, tags=key_update_item.tags, ) - + # Get team object and check team limits if team_id is provided team_obj: Optional[LiteLLM_TeamTableCachedObj] = None if update_key_request.team_id is not None: @@ -1568,18 +1575,16 @@ async def _process_single_key_update( user_api_key_cache=user_api_key_cache, check_db_only=True, ) - + if team_obj is not None and prisma_client is not None: await _check_team_key_limits( team_table=team_obj, data=update_key_request, prisma_client=prisma_client, ) - + # Validate team change if team is being changed - if is_different_team( - data=update_key_request, existing_key_row=existing_key_row - ): + if is_different_team(data=update_key_request, existing_key_row=existing_key_row): if llm_router is None: raise HTTPException( status_code=400, @@ -1590,9 +1595,7 @@ async def _process_single_key_update( if team_obj is None: raise HTTPException( status_code=500, - detail={ - "error": "Team object not found for team change validation" - }, + detail={"error": "Team object not found for team change validation"}, ) validate_key_team_change( key=existing_key_row, @@ -1600,31 +1603,29 @@ async def _process_single_key_update( change_initiated_by=user_api_key_dict, llm_router=llm_router, ) - + # Prepare update data non_default_values = await prepare_key_update_data( data=update_key_request, existing_key_row=existing_key_row ) - + # Update key in database if prisma_client is None: raise HTTPException( status_code=500, detail={"error": "Database not connected"}, ) - + _data = {**non_default_values, "token": key_update_item.key} - response = await prisma_client.update_data( - token=key_update_item.key, data=_data - ) - + response = await prisma_client.update_data(token=key_update_item.key, data=_data) + # Delete cache await _delete_cache_key_object( hashed_token=hash_token(key_update_item.key), user_api_key_cache=user_api_key_cache, proxy_logging_obj=proxy_logging_obj, ) - + # Trigger async hook asyncio.create_task( KeyManagementEventHooks.async_key_updated_hook( @@ -1635,19 +1636,19 @@ async def _process_single_key_update( litellm_changed_by=litellm_changed_by, ) ) - + if response is None: raise ValueError("Failed to update key got response = None") - + # Extract and format updated key info updated_key_info = response.get("data", {}) if hasattr(updated_key_info, "model_dump"): updated_key_info = updated_key_info.model_dump() elif hasattr(updated_key_info, "dict"): updated_key_info = updated_key_info.dict() - + updated_key_info.pop("token", None) - + return updated_key_info @@ -1740,7 +1741,9 @@ async def update_key_fn( if data.max_budget is not None and data.max_budget < 0: raise HTTPException( status_code=400, - detail={"error": f"max_budget cannot be negative. Received: {data.max_budget}"} + detail={ + "error": f"max_budget cannot be negative. Received: {data.max_budget}" + }, ) data_json: dict = data.model_dump(exclude_unset=True, exclude_none=True) @@ -1959,13 +1962,11 @@ async def bulk_update_keys( proxy_logging_obj, user_api_key_cache, ) - + if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value: raise HTTPException( status_code=403, - detail={ - "error": "Only proxy admins can perform bulk key updates" - }, + detail={"error": "Only proxy admins can perform bulk key updates"}, ) if prisma_client is None: @@ -2381,10 +2382,10 @@ async def info_key_fn( # if using pydantic v1 key_info = key_info.dict() key_info.pop("token") - + # Attach object_permission if object_permission_id is set key_info = await attach_object_permission_to_dict(key_info, prisma_client) - + return {"key": key, "info": key_info} except Exception as e: raise handle_exception_on_proxy(e) @@ -2509,7 +2510,9 @@ async def generate_key_helper_fn( # noqa: PLR0915 aliases_json = json.dumps(aliases) config_json = json.dumps(config) permissions_json = json.dumps(permissions) - router_settings_json = safe_dumps(router_settings) if router_settings is not None else safe_dumps({}) + router_settings_json = ( + safe_dumps(router_settings) if router_settings is not None else safe_dumps({}) + ) # Add model_rpm_limit and model_tpm_limit to metadata if model_rpm_limit is not None: @@ -2676,10 +2679,12 @@ async def generate_key_helper_fn( # noqa: PLR0915 ) key_data["created_at"] = getattr(create_key_response, "created_at", None) key_data["updated_at"] = getattr(create_key_response, "updated_at", None) - + # Deserialize router_settings from JSON string to dict for response router_settings_value = key_data.get("router_settings") - if router_settings_value is not None and isinstance(router_settings_value, str): + if router_settings_value is not None and isinstance( + router_settings_value, str + ): try: key_data["router_settings"] = yaml.safe_load(router_settings_value) except yaml.YAMLError: @@ -2762,27 +2767,27 @@ async def can_modify_verification_token( ) -> bool: """ Check if user has permission to modify (delete/regenerate) a verification token. - + Rules: - Proxy admin can modify any key - For team keys: only team admin or key owner can modify - For personal keys: only key owner can modify - + Args: key_info: The verification token to check user_api_key_cache: Cache for user API keys user_api_key_dict: The user making the request prisma_client: Prisma client for database access - + Returns: True if user can modify the key, False otherwise """ is_team_key = _is_team_key(data=key_info) - + # 1. Proxy admin can modify any key if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: return True - + # 2. For team keys: only team admin or key owner can modify if is_team_key and key_info.team_id is not None: # Get team object to check if user is team admin @@ -2792,34 +2797,35 @@ async def can_modify_verification_token( user_api_key_cache=user_api_key_cache, check_db_only=True, ) - + if team_table is None: return False - + # Check if user is team admin if _is_user_team_admin( user_api_key_dict=user_api_key_dict, team_obj=team_table, ): return True - + # Check if the key belongs to the user (they own it) - if key_info.user_id is not None and key_info.user_id == user_api_key_dict.user_id: + if ( + key_info.user_id is not None + and key_info.user_id == user_api_key_dict.user_id + ): return True - + # Not team admin and doesn't own the key return False - + # 3. For personal keys: only key owner can modify if key_info.user_id is not None and key_info.user_id == user_api_key_dict.user_id: return True - + # Default: deny return False - - async def delete_verification_tokens( tokens: List, user_api_key_cache: DualCache, @@ -2849,10 +2855,10 @@ async def delete_verification_tokens( try: if prisma_client: tokens = [_hash_token_if_needed(token=key) for key in tokens] - _keys_being_deleted: List[LiteLLM_VerificationToken] = ( - await prisma_client.db.litellm_verificationtoken.find_many( - where={"token": {"in": tokens}} - ) + _keys_being_deleted: List[ + LiteLLM_VerificationToken + ] = await prisma_client.db.litellm_verificationtoken.find_many( + where={"token": {"in": tokens}} ) if len(_keys_being_deleted) == 0: @@ -2952,11 +2958,24 @@ def _transform_verification_tokens_to_deleted_records( if org_id_value is not None: record["organization_id"] = org_id_value - for json_field in ["aliases", "config", "permissions", "metadata", "model_spend", "model_max_budget", "router_settings"]: + for json_field in [ + "aliases", + "config", + "permissions", + "metadata", + "model_spend", + "model_max_budget", + "router_settings", + ]: if json_field in record and record[json_field] is not None: record[json_field] = json.dumps(record[json_field]) - for rel_key in ("litellm_budget_table", "litellm_organization_table", "object_permission", "id"): + for rel_key in ( + "litellm_budget_table", + "litellm_organization_table", + "object_permission", + "id", + ): record.pop(rel_key, None) records.append(record) @@ -2971,9 +2990,7 @@ async def _save_deleted_verification_token_records( """Save deleted verification token records to the database.""" if not records: return - await prisma_client.db.litellm_deletedverificationtoken.create_many( - data=records - ) + await prisma_client.db.litellm_deletedverificationtoken.create_many(data=records) async def _persist_deleted_verification_tokens( @@ -3036,9 +3053,9 @@ async def _rotate_master_key( from litellm.proxy.proxy_server import proxy_config try: - models: Optional[List] = ( - await prisma_client.db.litellm_proxymodeltable.find_many() - ) + models: Optional[ + List + ] = await prisma_client.db.litellm_proxymodeltable.find_many() except Exception: models = None # 2. process model table @@ -3115,7 +3132,9 @@ async def _rotate_master_key( updated_patch=decrypted_cred, new_encryption_key=new_master_key, ) - credential_object_jsonified = jsonify_object(encrypted_cred.model_dump()) + credential_object_jsonified = jsonify_object( + encrypted_cred.model_dump() + ) await prisma_client.db.litellm_credentialstable.update( where={"credential_name": cred.credential_name}, data={ @@ -3427,7 +3446,9 @@ def _validate_reset_spend_value( if reset_to > current_spend: raise HTTPException( status_code=status.HTTP_400_BAD_REQUEST, - detail={"error": f"reset_to ({reset_to}) must be <= current spend ({current_spend})"}, + detail={ + "error": f"reset_to ({reset_to}) must be <= current spend ({current_spend})" + }, ) max_budget = key_in_db.max_budget @@ -3553,11 +3574,11 @@ async def validate_key_list_check( param="user_id", code=status.HTTP_403_FORBIDDEN, ) - complete_user_info_db_obj: Optional[BaseModel] = ( - await prisma_client.db.litellm_usertable.find_unique( - where={"user_id": user_api_key_dict.user_id}, - include={"organization_memberships": True}, - ) + complete_user_info_db_obj: Optional[ + BaseModel + ] = await prisma_client.db.litellm_usertable.find_unique( + where={"user_id": user_api_key_dict.user_id}, + include={"organization_memberships": True}, ) if complete_user_info_db_obj is None: @@ -3643,10 +3664,10 @@ async def get_admin_team_ids( if complete_user_info is None: return [] # Get all teams that user is an admin of - teams: Optional[List[BaseModel]] = ( - await prisma_client.db.litellm_teamtable.find_many( - where={"team_id": {"in": complete_user_info.teams}} - ) + teams: Optional[ + List[BaseModel] + ] = await prisma_client.db.litellm_teamtable.find_many( + where={"team_id": {"in": complete_user_info.teams}} ) if teams is None: return [] @@ -3691,8 +3712,12 @@ async def list_keys( description="Column to sort by (e.g. 'user_id', 'created_at', 'spend')", ), sort_order: str = Query(default="desc", description="Sort order ('asc' or 'desc')"), - expand: Optional[List[str]] = Query(None, description="Expand related objects (e.g. 'user')"), - status: Optional[str] = Query(None, description="Filter by status (e.g. 'deleted')"), + expand: Optional[List[str]] = Query( + None, description="Expand related objects (e.g. 'user')" + ), + status: Optional[str] = Query( + None, description="Filter by status (e.g. 'deleted')" + ), ) -> KeyListResponseObject: """ List all keys for a given user / team / organization. @@ -3784,7 +3809,9 @@ async def list_keys( message=getattr(e, "detail", f"error({str(e)})"), type=ProxyErrorTypes.internal_server_error, param=getattr(e, "param", "None"), - code=getattr(e, "status_code", fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR), + code=getattr( + e, "status_code", fastapi.status.HTTP_500_INTERNAL_SERVER_ERROR + ), ) elif isinstance(e, ProxyException): raise e @@ -4617,10 +4644,16 @@ def validate_model_max_budget(model_max_budget: Optional[Dict]) -> None: for _model, _budget_info in model_max_budget.items(): assert isinstance(_model, str) + # Normalize to dict (Pydantic may already parse nested values as BudgetConfig) + _info = ( + _budget_info.model_dump() + if hasattr(_budget_info, "model_dump") + else dict(_budget_info) + ) # /CRUD endpoints can pass budget_limit as a string, so we need to convert it to a float - if "budget_limit" in _budget_info: - _budget_info["budget_limit"] = float(_budget_info["budget_limit"]) - BudgetConfig(**_budget_info) + if "budget_limit" in _info: + _info["budget_limit"] = float(_info["budget_limit"]) + BudgetConfig(**_info) except Exception as e: raise ValueError( f"Invalid model_max_budget: {str(e)}. Example of valid model_max_budget: https://docs.litellm.ai/docs/proxy/users" diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py index 64f1ec24234..54b9e31a6da 100644 --- a/tests/proxy_unit_tests/test_proxy_utils.py +++ b/tests/proxy_unit_tests/test_proxy_utils.py @@ -22,7 +22,6 @@ from litellm.proxy.litellm_pre_call_utils import ( _get_dynamic_logging_metadata, add_litellm_data_to_request, ) -from litellm.types.utils import SupportedCacheControls @pytest.fixture @@ -496,9 +495,7 @@ def test_add_litellm_data_for_backend_llm_call( from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup - user_api_key_dict = UserAPIKeyAuth( - api_key="test_api_key", user_id="test_user_id", org_id="test_org_id" - ) + UserAPIKeyAuth(api_key="test_api_key", user_id="test_user_id", org_id="test_org_id") data = LiteLLMProxyRequestSetup.get_user_from_headers( headers=headers, @@ -1059,7 +1056,7 @@ def test_update_config_fields_default_internal_user_params(monkeypatch): }, }, } - updated_config = proxy_config._update_config_fields(**args) + proxy_config._update_config_fields(**args) assert litellm.default_internal_user_params == { "user_role": "proxy_admin", @@ -1320,6 +1317,61 @@ def test_litellm_verification_token_view_response_with_budget_table( ) +def test_litellm_verification_token_view_budget_does_not_override_key_model_max_budget(): + """ + When key has non-empty model_max_budget, budget's model_max_budget is NOT applied. + Regression test for per-model budget: only apply budget's model_max_budget when key's is empty. + """ + from litellm.proxy._types import LiteLLM_VerificationTokenView + + key_model_max_budget = {"gpt-4": {"max_budget": 50.0, "budget_duration": "1d"}} + args = { + "token": "sk-test-mock-token-303", + "key_name": "sk-...if_g", + "key_alias": None, + "soft_budget_cooldown": False, + "spend": 0.0, + "expires": None, + "models": [], + "aliases": {}, + "config": {}, + "user_id": None, + "team_id": "test", + "permissions": {}, + "max_parallel_requests": None, + "metadata": {}, + "blocked": None, + "tpm_limit": None, + "rpm_limit": None, + "max_budget": None, + "budget_duration": None, + "budget_reset_at": None, + "allowed_cache_controls": [], + "model_spend": {}, + "model_max_budget": key_model_max_budget, + "budget_id": "my-test-tier", + "created_at": "2024-12-26T02:28:52.615+00:00", + "updated_at": "2024-12-26T03:01:51.159+00:00", + "team_spend": None, + "team_max_budget": None, + "team_tpm_limit": None, + "team_rpm_limit": None, + "team_models": [], + "team_metadata": {}, + "team_blocked": False, + "team_alias": None, + "team_members_with_roles": [], + "team_member_spend": None, + "team_model_aliases": None, + "team_member": None, + "litellm_budget_table_model_max_budget": { + "gpt-4o": {"max_budget": 100.0, "budget_duration": "1d"} + }, + } + resp = LiteLLM_VerificationTokenView(**args) + assert resp.model_max_budget == key_model_max_budget + + def test_is_allowed_to_make_key_request(): from litellm.proxy._types import LitellmUserRoles from litellm.proxy.management_endpoints.key_management_endpoints import ( @@ -1381,13 +1433,6 @@ def test_get_model_group_info(): assert len(model_list) == 1 -import asyncio -import json -from unittest.mock import AsyncMock, patch - -import pytest - - @pytest.fixture def mock_team_data(): return [ @@ -1444,7 +1489,6 @@ async def test_get_user_info_for_proxy_admin(mock_team_data, mock_key_data): "litellm.proxy.proxy_server.prisma_client", MockPrismaClientDB(mock_team_data, mock_key_data), ): - from litellm.proxy.management_endpoints.internal_user_endpoints import ( _get_user_info_for_proxy_admin, ) @@ -1558,9 +1602,6 @@ def test_update_key_budget_with_temp_budget_increase(): assert _update_key_budget_with_temp_budget_increase(valid_token).max_budget == 200 -from unittest.mock import AsyncMock, MagicMock - - @pytest.mark.asyncio async def test_health_check_not_called_when_disabled(monkeypatch): from litellm.proxy.proxy_server import ProxyStartupEvent @@ -1603,18 +1644,12 @@ async def test_health_check_not_called_when_disabled(monkeypatch): }, ) def test_custom_openapi(mock_get_openapi_schema): - from litellm.proxy.proxy_server import app, custom_openapi + from litellm.proxy.proxy_server import custom_openapi openapi_schema = custom_openapi() assert openapi_schema is not None -import asyncio -from datetime import timedelta -from unittest.mock import AsyncMock, MagicMock - -import pytest - from litellm.proxy.utils import ProxyUpdateSpend @@ -1639,6 +1674,7 @@ async def test_end_user_transactions_reset(): async def test_spend_logs_cleanup_after_error(): # Setup test data import asyncio + mock_client = MagicMock() mock_client.spend_log_transactions = [ {"id": 1, "amount": 10.0}, @@ -1826,7 +1862,7 @@ def test_provider_specific_header_in_request(custom_llm_provider, expected_resul client = HTTPHandler() with patch.object(client, "post", return_value=MagicMock()) as mock_post: try: - resp = litellm.completion( + litellm.completion( model="anthropic/claude-3-5-sonnet-v2@20241022", messages=[{"role": "user", "content": "Hello world"}], provider_specific_header=ProviderSpecificHeader( @@ -2063,7 +2099,7 @@ async def test_post_call_failure_hook_auth_error_key_info_route(): Test that post_call_failure_hook does NOT call _handle_logging_proxy_only_error when we get an auth error from /key/info route (since it's not an LLM API route). """ - from unittest.mock import AsyncMock, Mock, patch + from unittest.mock import AsyncMock, patch from fastapi import HTTPException @@ -2117,7 +2153,7 @@ async def test_post_call_failure_hook_auth_error_llm_api_route(): Test that post_call_failure_hook DOES call _handle_logging_proxy_only_error when we get an auth error from /v1/chat/completions route (since it is an LLM API route). """ - from unittest.mock import AsyncMock, Mock, patch + from unittest.mock import AsyncMock, patch from fastapi import HTTPException @@ -2182,27 +2218,27 @@ async def test_during_call_hook_parallel_execution(): cache = DualCache() proxy_logging = ProxyLogging(user_api_key_cache=cache) execution_order = [] - + class TestGuardrail(CustomGuardrail): def __init__(self, name): super().__init__( guardrail_name=name, event_hook=GuardrailEventHooks.during_call, - default_on=True + default_on=True, ) self.name = name - + async def async_moderation_hook(self, data, user_api_key_dict, call_type): execution_order.append(f"{self.name}_start") await asyncio.sleep(0.1) execution_order.append(f"{self.name}_end") return data - + original_callbacks = litellm.callbacks.copy() if litellm.callbacks else [] - + try: litellm.callbacks = [TestGuardrail(f"g{i}") for i in range(3)] - + start_time = asyncio.get_event_loop().time() result = await proxy_logging.during_call_hook( data={"model": "gpt-4", "messages": [{"role": "user", "content": "test"}]}, @@ -2210,14 +2246,22 @@ async def test_during_call_hook_parallel_execution(): call_type="completion", ) execution_time = asyncio.get_event_loop().time() - start_time - + # Verify parallel execution: all start before any end - first_end_idx = next(i for i, item in enumerate(execution_order) if "end" in item) - starts_before_end = sum(1 for item in execution_order[:first_end_idx] if "start" in item) - assert starts_before_end == 3, f"Expected 3 starts before first end, got {starts_before_end}" - + first_end_idx = next( + i for i, item in enumerate(execution_order) if "end" in item + ) + starts_before_end = sum( + 1 for item in execution_order[:first_end_idx] if "start" in item + ) + assert ( + starts_before_end == 3 + ), f"Expected 3 starts before first end, got {starts_before_end}" + # Verify timing: parallel ~0.1s vs sequential ~0.3s - assert execution_time < 0.2, f"Parallel execution took {execution_time}s, expected < 0.2s" + assert ( + execution_time < 0.2 + ), f"Parallel execution took {execution_time}s, expected < 0.2s" assert result["model"] == "gpt-4" finally: litellm.callbacks = original_callbacks @@ -2235,30 +2279,35 @@ async def test_during_call_hook_parallel_execution_with_error(): cache = DualCache() proxy_logging = ProxyLogging(user_api_key_cache=cache) - + class FailingGuardrail(CustomGuardrail): def __init__(self): super().__init__( guardrail_name="failing_guardrail", event_hook=GuardrailEventHooks.during_call, - default_on=True + default_on=True, ) - + async def async_moderation_hook(self, data, user_api_key_dict, call_type): raise ValueError("Guardrail violation detected!") - + original_callbacks = litellm.callbacks.copy() if litellm.callbacks else [] - + try: litellm.callbacks = [FailingGuardrail()] - + with pytest.raises(ValueError) as exc_info: await proxy_logging.during_call_hook( - data={"model": "gpt-4", "messages": [{"role": "user", "content": "test"}]}, - user_api_key_dict=UserAPIKeyAuth(api_key="test_key", user_id="test_user"), + data={ + "model": "gpt-4", + "messages": [{"role": "user", "content": "test"}], + }, + user_api_key_dict=UserAPIKeyAuth( + api_key="test_key", user_id="test_user" + ), call_type="completion", ) - + assert "Guardrail violation detected!" in str(exc_info.value) finally: - litellm.callbacks = original_callbacks \ No newline at end of file + litellm.callbacks = original_callbacks diff --git a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py index fc8373a1746..352db384c88 100644 --- a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py +++ b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py @@ -1,30 +1,20 @@ -import json import os import sys -from datetime import datetime -from unittest.mock import AsyncMock +from unittest.mock import AsyncMock, patch sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system-path -from datetime import datetime as dt_object -import time -import pytest -import litellm -import json -from litellm.types.utils import BudgetConfig as GenericBudgetInfo -import os -import sys -from datetime import datetime -from unittest.mock import AsyncMock, patch import pytest + +import litellm from litellm.caching.caching import DualCache from litellm.proxy.hooks.model_max_budget_limiter import ( _PROXY_VirtualKeyModelMaxBudgetLimiter, ) from litellm.proxy._types import UserAPIKeyAuth -import litellm +from litellm.types.utils import BudgetConfig as GenericBudgetInfo # Test class setup @@ -123,3 +113,48 @@ async def test_get_virtual_key_spend_for_model(budget_limiter): key_budget_config=budget_config, ) assert spend == 50.0 + + +@pytest.mark.asyncio +async def test_async_log_success_event_uses_per_model_budget_duration(budget_limiter): + """ + async_log_success_event must use the per-model budget_duration for the cache key + so spend is tracked per model correctly. Regression test for per-model budget implementation. + """ + from litellm.proxy.hooks.model_max_budget_limiter import ( + VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX, + ) + + virtual_key = "test-key-hash" + model = "gpt-4" + budget_duration = "1d" + user_api_key_model_max_budget = { + model: {"budget_limit": 100.0, "time_period": budget_duration}, + } + kwargs = { + "standard_logging_object": { + "response_cost": 0.05, + "model": model, + "metadata": {"user_api_key_hash": virtual_key}, + }, + "litellm_params": { + "metadata": { + "user_api_key_model_max_budget": user_api_key_model_max_budget + }, + }, + } + with patch.object( + budget_limiter, + "_increment_spend_for_key", + new_callable=AsyncMock, + ) as mock_increment: + await budget_limiter.async_log_success_event( + kwargs, response_obj=None, start_time=None, end_time=None + ) + mock_increment.assert_awaited_once() + call_kwargs = mock_increment.call_args.kwargs + spend_key = call_kwargs["spend_key"] + assert spend_key == ( + f"{VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX}:{virtual_key}:{model}:{budget_duration}" + ) + assert call_kwargs["response_cost"] == 0.05 diff --git a/tests/test_litellm/proxy/management_endpoints/test_budget_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_budget_endpoints.py index d5c3ecae7d6..d8c505223d9 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_budget_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_budget_endpoints.py @@ -11,7 +11,6 @@ import litellm.proxy.proxy_server as ps from litellm.proxy.proxy_server import app from litellm.proxy._types import UserAPIKeyAuth, LitellmUserRoles, CommonProxyErrors -import litellm.proxy.management_endpoints.budget_management_endpoints as bm sys.path.insert( 0, os.path.abspath("../../../") @@ -22,13 +21,12 @@ sys.path.insert( def client_and_mocks(monkeypatch): # Setup MagicMock Prisma mock_prisma = MagicMock() - mock_table = MagicMock() mock_table.create = AsyncMock(side_effect=lambda *, data: data) mock_table.update = AsyncMock(side_effect=lambda *, where, data: {**where, **data}) mock_prisma.db = types.SimpleNamespace( - litellm_budgettable = mock_table, - litellm_dailyspend = mock_table, + litellm_budgettable=mock_table, + litellm_dailyspend=mock_table, ) # Monkeypatch Mocked Prisma client into the server module @@ -79,6 +77,7 @@ async def test_new_budget_db_not_connected(client_and_mocks, monkeypatch): # override the prisma_client that the handler imports at runtime import litellm.proxy.proxy_server as ps + monkeypatch.setattr(ps, "prisma_client", None) # Call /budget/new endpoint @@ -123,6 +122,7 @@ async def test_update_budget_db_not_connected(client_and_mocks, monkeypatch): # override the prisma_client that the handler imports at runtime import litellm.proxy.proxy_server as ps + monkeypatch.setattr(ps, "prisma_client", None) payload = {"budget_id": "any", "max_budget": 1.0} @@ -136,7 +136,7 @@ async def test_update_budget_db_not_connected(client_and_mocks, monkeypatch): async def test_update_budget_allows_null_max_budget(client_and_mocks): """ Test that /budget/update allows setting max_budget to null. - + Previously, using exclude_none=True would drop null values, making it impossible to remove a budget limit. With exclude_unset=True, explicitly setting max_budget to null should include it in the update. @@ -144,11 +144,11 @@ async def test_update_budget_allows_null_max_budget(client_and_mocks): client, _, mock_table = client_and_mocks captured_data = {} - + async def capture_update(*, where, data): captured_data.update(data) return {**where, **data} - + mock_table.update = AsyncMock(side_effect=capture_update) payload = { @@ -159,9 +159,11 @@ async def test_update_budget_allows_null_max_budget(client_and_mocks): assert resp.status_code == 200, resp.text # Verify that max_budget=None was included in the update data - assert "max_budget" in captured_data, "max_budget should be included when explicitly set to null" + assert ( + "max_budget" in captured_data + ), "max_budget should be included when explicitly set to null" assert captured_data["max_budget"] is None, "max_budget should be None" - + mock_table.update.assert_awaited_once() @@ -169,7 +171,7 @@ async def test_update_budget_allows_null_max_budget(client_and_mocks): async def test_new_budget_negative_max_budget(client_and_mocks): """ Test that /budget/new rejects negative max_budget values. - + This prevents the issue where negative budgets would always trigger budget exceeded errors. """ @@ -181,7 +183,7 @@ async def test_new_budget_negative_max_budget(client_and_mocks): } resp = client.post("/budget/new", json=payload) assert resp.status_code == 400, resp.text - + detail = resp.json()["detail"] assert "max_budget cannot be negative" in str(detail) @@ -199,7 +201,7 @@ async def test_new_budget_negative_soft_budget(client_and_mocks): } resp = client.post("/budget/new", json=payload) assert resp.status_code == 400, resp.text - + detail = resp.json()["detail"] assert "soft_budget cannot be negative" in str(detail) @@ -217,7 +219,7 @@ async def test_update_budget_negative_max_budget(client_and_mocks): } resp = client.post("/budget/update", json=payload) assert resp.status_code == 400, resp.text - + detail = resp.json()["detail"] assert "max_budget cannot be negative" in str(detail) @@ -235,6 +237,30 @@ async def test_update_budget_negative_soft_budget(client_and_mocks): } resp = client.post("/budget/update", json=payload) assert resp.status_code == 400, resp.text - + detail = resp.json()["detail"] assert "soft_budget cannot be negative" in str(detail) + + +@pytest.mark.asyncio +async def test_new_budget_invalid_model_max_budget(client_and_mocks, monkeypatch): + """ + Test that /budget/new validates model_max_budget and returns 400 for invalid structure. + Per-model budget implementation: validate_model_max_budget is called in new_budget. + """ + import litellm.proxy.proxy_server as ps + + monkeypatch.setattr(ps, "premium_user", True) + + client, _, _ = client_and_mocks + + payload = { + "budget_id": "budget_invalid_mmb", + "max_budget": 10.0, + "model_max_budget": {"gpt-4": "not-a-dict"}, + } + resp = client.post("/budget/new", json=payload) + # Pydantic may reject invalid structure with 422 before our validator runs + assert resp.status_code in (400, 422), resp.text + detail = resp.json()["detail"] + assert "model_max_budget" in str(detail) or "dictionary" in str(detail).lower() From e24ea2897a9b8217eb89c004782afdb34db85022 Mon Sep 17 00:00:00 2001 From: Varun Chawla <34209028+veeceey@users.noreply.github.com> Date: Sat, 7 Feb 2026 22:22:12 -0800 Subject: [PATCH 10/50] fix: empty guardrails/policies arrays should not trigger enterprise license check (#20567) * fix: empty guardrails/policies arrays should not trigger enterprise license check (#20304) The UI sends empty arrays for enterprise-only fields (guardrails, policies, logging) even when the user has not configured these features. The backend `is not None` check treated `[]` as a truthy intent to use the feature, falsely requiring an enterprise license for basic team operations. Backend: Add `and updated_kv[field] != [] and updated_kv[field] != {}` guards in `_update_metadata_fields` so empty collections are skipped. UI: Conditionally omit guardrails, logging, and policies from the payload when empty instead of defaulting to `[]`. Fixes #20304 * fix: allow clearing fields with empty collections while skipping enterprise check Address PR review feedback: 1. Move the empty-collection guard into _update_metadata_field (singular) so that empty lists/dicts skip only the premium license check but still get written into metadata. This lets users intentionally clear a previously-set field (e.g. guardrails: []) without being blocked, while the UI's default empty arrays still don't trigger a false enterprise error. 2. Remove sys.path hack from test file; use standard imports that work with pytest discovery. 3. Add tests verifying that empty collections are moved into metadata (field clearing works) even though they bypass the premium check. Fixes #20304 --- .../management_endpoints/common_utils.py | 9 +- .../management_endpoints/test_common_utils.py | 162 ++++++++++++++++++ .../src/components/team/team_info.tsx | 6 +- 3 files changed, 173 insertions(+), 4 deletions(-) create mode 100644 tests/test_litellm/proxy/management_endpoints/test_common_utils.py diff --git a/litellm/proxy/management_endpoints/common_utils.py b/litellm/proxy/management_endpoints/common_utils.py index 8f7dd4f8dfa..24a41a2361b 100644 --- a/litellm/proxy/management_endpoints/common_utils.py +++ b/litellm/proxy/management_endpoints/common_utils.py @@ -216,7 +216,14 @@ def _update_metadata_field(updated_kv: dict, field_name: str) -> None: field_name: Name of the metadata field being updated """ if field_name in LiteLLM_ManagementEndpoint_MetadataFields_Premium: - _premium_user_check() + value = updated_kv.get(field_name) + # Skip the premium check for empty collections ([] or {}). + # The UI sends these as defaults even when the user hasn't configured + # any enterprise features (see issue #20304). However, we still + # proceed with the update so that users can intentionally clear a + # previously-set field by sending an empty list/dict. + if value is not None and value != [] and value != {}: + _premium_user_check() if field_name in updated_kv and updated_kv[field_name] is not None: # remove field from updated_kv diff --git a/tests/test_litellm/proxy/management_endpoints/test_common_utils.py b/tests/test_litellm/proxy/management_endpoints/test_common_utils.py new file mode 100644 index 00000000000..b372476c3d6 --- /dev/null +++ b/tests/test_litellm/proxy/management_endpoints/test_common_utils.py @@ -0,0 +1,162 @@ +""" +Tests for litellm/proxy/management_endpoints/common_utils.py + +Covers the fix for GitHub issue #20304: +Empty guardrails/policies arrays sent by the UI should NOT trigger the +enterprise (premium) license check, but should still be applied so that +users can intentionally clear previously-set fields. +""" + +from unittest.mock import patch + +from litellm.proxy.management_endpoints.common_utils import ( + _update_metadata_fields, +) + + +class TestUpdateMetadataFieldsEmptyCollections: + """ + Regression tests for issue #20304. + + The UI sends empty arrays (`[]`) for enterprise-only fields like + guardrails, policies, and logging even when the user hasn't configured + these features. The backend must not treat empty collections as an + intent to use the feature, and therefore must not trigger the premium + license check. + + However, empty collections must still be written into metadata so that + users can intentionally clear a previously-set field (e.g. removing all + guardrails by sending `guardrails: []`). + """ + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_empty_list_does_not_trigger_premium_check(self, mock_premium_check): + """Empty lists for premium fields must not trigger the premium check.""" + updated_kv = { + "team_id": "test-team", + "guardrails": [], + "policies": [], + "logging": [], + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_not_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_empty_list_still_updates_metadata(self, mock_premium_check): + """ + Empty lists must still be moved into metadata so users can clear + previously-set fields (e.g. remove all guardrails). + """ + updated_kv = { + "team_id": "test-team", + "guardrails": [], + "policies": [], + } + _update_metadata_fields(updated_kv=updated_kv) + # The fields should have been moved into metadata + assert "guardrails" not in updated_kv, ( + "guardrails should be popped from top-level" + ) + assert "policies" not in updated_kv, ( + "policies should be popped from top-level" + ) + assert updated_kv["metadata"]["guardrails"] == [] + assert updated_kv["metadata"]["policies"] == [] + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_empty_dict_does_not_trigger_premium_check(self, mock_premium_check): + """Empty dicts for premium fields must not trigger the premium check.""" + updated_kv = { + "team_id": "test-team", + "secret_manager_settings": {}, + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_not_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_empty_dict_still_updates_metadata(self, mock_premium_check): + """ + Empty dicts must still be moved into metadata so users can clear + previously-set fields. + """ + updated_kv = { + "team_id": "test-team", + "secret_manager_settings": {}, + } + _update_metadata_fields(updated_kv=updated_kv) + assert "secret_manager_settings" not in updated_kv, ( + "secret_manager_settings should be popped from top-level" + ) + assert updated_kv["metadata"]["secret_manager_settings"] == {} + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_none_value_does_not_trigger_premium_check(self, mock_premium_check): + """None values for premium fields should be silently ignored.""" + updated_kv = { + "team_id": "test-team", + "guardrails": None, + "policies": None, + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_not_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_absent_fields_do_not_trigger_premium_check(self, mock_premium_check): + """Fields not present in the dict should not trigger premium check.""" + updated_kv = { + "team_id": "test-team", + "team_alias": "example-team", + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_not_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_non_empty_list_triggers_premium_check(self, mock_premium_check): + """Non-empty lists for premium fields should trigger the premium check.""" + updated_kv = { + "team_id": "test-team", + "guardrails": ["my-guardrail"], + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_non_empty_value_triggers_premium_check(self, mock_premium_check): + """Non-empty string values for premium fields should trigger the premium check.""" + updated_kv = { + "team_id": "test-team", + "tags": ["production"], + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_called() + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_non_empty_list_updates_metadata(self, mock_premium_check): + """Non-empty lists should be moved into metadata.""" + updated_kv = { + "team_id": "test-team", + "guardrails": ["my-guardrail"], + } + _update_metadata_fields(updated_kv=updated_kv) + assert "guardrails" not in updated_kv + assert updated_kv["metadata"]["guardrails"] == ["my-guardrail"] + + @patch("litellm.proxy.management_endpoints.common_utils._premium_user_check") + def test_ui_typical_payload_does_not_trigger_premium_check(self, mock_premium_check): + """ + Simulate the exact payload the UI sends when no enterprise features + are configured. This must NOT trigger the premium check. + """ + # This is the payload structure the UI sends (from issue #20304) + updated_kv = { + "team_id": "67848772-1a8b-4343-938c-17e60f1db860", + "team_alias": "example-team", + "models": ["gpt-4"], + "metadata": { + "guardrails": [], + "logging": [], + }, + "policies": [], + } + _update_metadata_fields(updated_kv=updated_kv) + mock_premium_check.assert_not_called() diff --git a/ui/litellm-dashboard/src/components/team/team_info.tsx b/ui/litellm-dashboard/src/components/team/team_info.tsx index 35f5b87e071..014f8fb9010 100644 --- a/ui/litellm-dashboard/src/components/team/team_info.tsx +++ b/ui/litellm-dashboard/src/components/team/team_info.tsx @@ -465,8 +465,8 @@ const TeamInfoView: React.FC = ({ budget_duration: values.budget_duration, metadata: { ...parsedMetadata, - guardrails: values.guardrails || [], - logging: values.logging_settings || [], + ...(values.guardrails?.length > 0 ? { guardrails: values.guardrails } : {}), + ...(values.logging_settings?.length > 0 ? { logging: values.logging_settings } : {}), disable_global_guardrails: values.disable_global_guardrails || false, soft_budget_alerting_emails: typeof values.soft_budget_alerting_emails === "string" @@ -477,7 +477,7 @@ const TeamInfoView: React.FC = ({ : values.soft_budget_alerting_emails || [], ...(secretManagerSettings !== undefined ? { secret_manager_settings: secretManagerSettings } : {}), }, - policies: values.policies || [], + ...(values.policies?.length > 0 ? { policies: values.policies } : {}), organization_id: values.organization_id, }; From 3b043ee8bfebe29b9f9071e658f209087042f5f0 Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Sun, 8 Feb 2026 11:53:01 +0530 Subject: [PATCH 11/50] fix critical CVE vulnerabliltes (#20683) --- .dockerignore | 2 +- Dockerfile | 31 ++++++++++++++++++- ci_cd/security_scans.sh | 5 +-- docker/Dockerfile.custom_ui | 13 +++++++- docker/Dockerfile.database | 29 ++++++++++++++--- docker/Dockerfile.dev | 27 +++++++++++++++- docker/Dockerfile.non_root | 29 ++++++++++++++--- docs/my-website/package.json | 2 ++ litellm-js/spend-logs/package.json | 4 ++- package.json | 4 ++- requirements.txt | 5 +++ tests/proxy_admin_ui_tests/package.json | 4 ++- .../ui_unit_tests/package.json | 4 ++- ui/litellm-dashboard/package.json | 2 ++ 14 files changed, 141 insertions(+), 20 deletions(-) diff --git a/.dockerignore b/.dockerignore index 76e31546c2f..a487d2a859a 100644 --- a/.dockerignore +++ b/.dockerignore @@ -48,7 +48,7 @@ dist/ build/ *.egg-info/ .DS_Store -node_modules/ +**/node_modules *.log .env .env.local diff --git a/Dockerfile b/Dockerfile index 717ec2bcb77..5e93a0c627e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -49,7 +49,22 @@ USER root # Install runtime dependencies (libsndfile needed for audio processing on ARM64) RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip libsndfile && \ - npm install -g npm@latest tar@latest + npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \ + # SECURITY FIX: npm bundles tar, glob, and brace-expansion at multiple nested + # levels inside its dependency tree. `npm install -g ` only creates a + # SEPARATE global package, it does NOT replace npm's internal copies. + # We must find and replace EVERY copy inside npm's directory. + GLOBAL="$(npm root -g)" && \ + find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done && \ + npm cache clean --force WORKDIR /app # Copy the current directory contents into the container at /app @@ -71,6 +86,20 @@ RUN NODEJS_WHEEL_NODE=$(find /usr/lib -path "*/nodejs_wheel/bin/node" 2>/dev/nul RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \ find /usr/lib -type d -path "*/tornado/test" -delete +# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete +# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/. +# Patch every copy of tar, glob, and brace-expansion inside that tree. +RUN GLOBAL="$(npm root -g)" && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done + # Install semantic_router and aurelio-sdk using script # Convert Windows line endings to Unix and make executable RUN sed -i 's/\r$//' docker/install_auto_router.sh && chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh diff --git a/ci_cd/security_scans.sh b/ci_cd/security_scans.sh index 770610c2a3b..3ffa13c444f 100755 --- a/ci_cd/security_scans.sh +++ b/ci_cd/security_scans.sh @@ -155,10 +155,7 @@ run_grype_scans() { "CVE-2025-12781" # No fix available yet "CVE-2025-11468" # No fix available yet "CVE-2026-1299" # Python 3.13 email module header injection - not applicable, LiteLLM doesn't use BytesGenerator for email serialization - "GHSA-7h2j-956f-4vf2" # @isaacs/brace-expansion ReDoS - npm tooling dependency, not used in application runtime - "GHSA-hx9q-6w63-j58v" # orjson deep recursion - no fix available yet - "GHSA-8qq5-rm4j-mr97" # node-tar symlink poisoning - npm tooling dependency, tar CLI not exposed in application code - "GHSA-29xp-372q-xqph" # node-tar race condition - npm tooling dependency, tar CLI not exposed in application code + "CVE-2026-0775" # npm cli incorrect permission assignment - no fix available yet, npm is only used at build/prisma-generate time ) # Build JSON array of allowlisted CVE IDs for jq diff --git a/docker/Dockerfile.custom_ui b/docker/Dockerfile.custom_ui index 57926bcd170..177d7b7b12a 100644 --- a/docker/Dockerfile.custom_ui +++ b/docker/Dockerfile.custom_ui @@ -6,7 +6,18 @@ WORKDIR /app # Install Node.js and npm (adjust version as needed) RUN apt-get update && apt-get install -y nodejs npm && \ - npm install -g npm@latest tar@latest + npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \ + GLOBAL="$(npm root -g)" && \ + find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done && \ + npm cache clean --force # Copy the UI source into the container COPY ./ui/litellm-dashboard /app/ui/litellm-dashboard diff --git a/docker/Dockerfile.database b/docker/Dockerfile.database index ecbe76446f6..a6fcd98ab6d 100644 --- a/docker/Dockerfile.database +++ b/docker/Dockerfile.database @@ -50,7 +50,18 @@ USER root # Install runtime dependencies RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip libsndfile && \ - npm install -g npm@latest tar@latest + npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \ + GLOBAL="$(npm root -g)" && \ + find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done && \ + npm cache clean --force WORKDIR /app # Copy the current directory contents into the container at /app @@ -64,9 +75,19 @@ COPY --from=builder /wheels/ /wheels/ # Install the built wheel using pip; again using a wildcard if it's the only file RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels -# Replace the nodejs-wheel-binaries bundled node with the system node (fixes CVE-2025-55130) -RUN NODEJS_WHEEL_NODE=$(find /usr/lib -path "*/nodejs_wheel/bin/node" 2>/dev/null) && \ - if [ -n "$NODEJS_WHEEL_NODE" ]; then cp /usr/bin/node "$NODEJS_WHEEL_NODE"; fi +# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete +# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/. +# Patch every copy of tar, glob, and brace-expansion inside that tree. +RUN GLOBAL="$(npm root -g)" && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done # Install semantic_router and aurelio-sdk using script # Convert Windows line endings to Unix and make executable diff --git a/docker/Dockerfile.dev b/docker/Dockerfile.dev index ae557d4647f..bc1d22d5e05 100644 --- a/docker/Dockerfile.dev +++ b/docker/Dockerfile.dev @@ -62,7 +62,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ nodejs \ npm \ && rm -rf /var/lib/apt/lists/* \ - && npm install -g npm@latest tar@latest + && npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 \ + && GLOBAL="$(npm root -g)" \ + && find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done \ + && find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done \ + && find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done \ + && npm cache clean --force WORKDIR /app @@ -80,6 +91,20 @@ RUN pip install --no-cache-dir *.whl /wheels/* --no-index --find-links=/wheels/ rm -f *.whl && \ rm -rf /wheels +# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete +# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/. +# Patch every copy of tar, glob, and brace-expansion inside that tree. +RUN GLOBAL="$(npm root -g)" && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done + # Generate prisma client and set permissions # Convert Windows line endings to Unix for entrypoint scripts RUN prisma generate && \ diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root index 4b09755ed7d..64126bb0292 100644 --- a/docker/Dockerfile.non_root +++ b/docker/Dockerfile.non_root @@ -104,7 +104,18 @@ RUN for i in 1 2 3; do \ && for i in 1 2 3; do \ apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \ done \ - && npm install -g npm@latest tar@latest + && npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 \ + && GLOBAL="$(npm root -g)" \ + && find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done \ + && find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done \ + && find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done \ + && npm cache clean --force # Copy artifacts from builder COPY --from=builder /app/requirements.txt /app/requirements.txt @@ -146,9 +157,19 @@ RUN pip install --no-index --find-links=/wheels/ -r requirements.txt && \ fi; \ fi -# Replace the nodejs-wheel-binaries bundled node with the system node (fixes CVE-2025-55130) -RUN NODEJS_WHEEL_NODE=$(find /usr/lib -path "*/nodejs_wheel/bin/node" 2>/dev/null) && \ - if [ -n "$NODEJS_WHEEL_NODE" ]; then cp /usr/bin/node "$NODEJS_WHEEL_NODE"; fi +# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete +# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/. +# Patch every copy of tar, glob, and brace-expansion inside that tree. +RUN GLOBAL="$(npm root -g)" && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \ + done && \ + find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \ + rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \ + done # Permissions, cleanup, and Prisma prep # Convert Windows line endings to Unix for entrypoint scripts diff --git a/docs/my-website/package.json b/docs/my-website/package.json index 4c3db680565..4af7a168f83 100644 --- a/docs/my-website/package.json +++ b/docs/my-website/package.json @@ -61,6 +61,8 @@ "mermaid": ">=11.10.0", "gray-matter": "4.0.3", "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1", "node-forge": ">=1.3.2", "mdast-util-to-hast": ">=13.2.1", "lodash-es": ">=4.17.23" diff --git a/litellm-js/spend-logs/package.json b/litellm-js/spend-logs/package.json index 9c1c2d4f6dc..67292567145 100644 --- a/litellm-js/spend-logs/package.json +++ b/litellm-js/spend-logs/package.json @@ -11,6 +11,8 @@ "tsx": "^4.7.1" }, "overrides": { - "glob": ">=11.1.0" + "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1" } } diff --git a/package.json b/package.json index 7f90fd0aeb9..ab9e15f46a7 100644 --- a/package.json +++ b/package.json @@ -11,6 +11,8 @@ "jest": "^29.7.0" }, "overrides": { - "glob": ">=11.1.0" + "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1" } } diff --git a/requirements.txt b/requirements.txt index 1f21cc62bc6..f680de120c5 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,4 +1,9 @@ # LITELLM PROXY DEPENDENCIES # +# Security: explicit pins for transitive deps (CVE fixes) +urllib3>=2.6.0 # CVE-2025-66471, CVE-2025-66418, CVE-2026-21441 +tornado>=6.5.3 # CVE-2025-67725, CVE-2025-67726, CVE-2025-67724 +filelock>=3.20.1 # CVE-2025-68146 + anyio==4.8.0 # openai + http req. httpx==0.28.1 openai==2.9.0 # openai req. diff --git a/tests/proxy_admin_ui_tests/package.json b/tests/proxy_admin_ui_tests/package.json index cbd25be8816..48de2c1dba2 100644 --- a/tests/proxy_admin_ui_tests/package.json +++ b/tests/proxy_admin_ui_tests/package.json @@ -12,6 +12,8 @@ "@types/node": "^22.5.5" }, "overrides": { - "glob": ">=11.1.0" + "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1" } } diff --git a/tests/proxy_admin_ui_tests/ui_unit_tests/package.json b/tests/proxy_admin_ui_tests/ui_unit_tests/package.json index 7d82ee2e1a6..4c7d7addf0e 100644 --- a/tests/proxy_admin_ui_tests/ui_unit_tests/package.json +++ b/tests/proxy_admin_ui_tests/ui_unit_tests/package.json @@ -24,6 +24,8 @@ "react-dom": "^18.2.0" }, "overrides": { - "glob": ">=11.1.0" + "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1" } } \ No newline at end of file diff --git a/ui/litellm-dashboard/package.json b/ui/litellm-dashboard/package.json index efb11fec386..76ac97f008c 100644 --- a/ui/litellm-dashboard/package.json +++ b/ui/litellm-dashboard/package.json @@ -84,6 +84,8 @@ "mermaid": ">=11.10.0", "js-yaml": ">=4.1.1", "glob": ">=11.1.0", + "tar": ">=7.5.7", + "@isaacs/brace-expansion": ">=5.0.1", "node-forge": ">=1.3.2", "lodash-es": ">=4.17.23", "lodash": ">=4.17.23" From 7f93ff9e83ef18510000ce8a72c09cf24172b8e0 Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Sun, 8 Feb 2026 12:12:19 +0530 Subject: [PATCH 12/50] fix: add hook to handle db case (#20635) --- litellm/integrations/datadog/datadog.py | 113 ++++++++++++++++++++++-- 1 file changed, 107 insertions(+), 6 deletions(-) diff --git a/litellm/integrations/datadog/datadog.py b/litellm/integrations/datadog/datadog.py index 127b0e53fa8..64e0b26a8e7 100644 --- a/litellm/integrations/datadog/datadog.py +++ b/litellm/integrations/datadog/datadog.py @@ -45,7 +45,14 @@ from litellm.llms.custom_httpx.http_handler import ( httpxSpecialProvider, ) from litellm.types.integrations.base_health_check import IntegrationHealthCheckStatus -from litellm.types.integrations.datadog import * +from litellm.types.integrations.datadog import ( + DD_ERRORS, + DD_MAX_BATCH_SIZE, + DataDogStatus, + DatadogInitParams, + DatadogPayload, + DatadogProxyFailureHookJsonMessage, +) from litellm.types.services import ServiceLoggerPayload, ServiceTypes from litellm.types.utils import StandardLoggingPayload @@ -85,12 +92,14 @@ class DataDogLogger( """ try: verbose_logger.debug("Datadog: in init datadog logger") - + self.is_mock_mode = should_use_datadog_mock() - + if self.is_mock_mode: create_mock_datadog_client() - verbose_logger.debug("[DATADOG MOCK] Datadog logger initialized in mock mode") + verbose_logger.debug( + "[DATADOG MOCK] Datadog logger initialized in mock mode" + ) ######################################################### # Handle datadog_params set as litellm.datadog_params @@ -209,6 +218,96 @@ class DataDogLogger( ) pass + async def async_post_call_failure_hook( + self, + request_data: dict, + original_exception: Exception, + user_api_key_dict: Any, + traceback_str: Optional[str] = None, + ) -> Optional[Any]: + """ + Log proxy-level failures (e.g. 401 auth, DB connection errors) to Datadog. + + Ensures failures that occur before or outside the LLM completion flow + (e.g. ConnectError during auth when DB is down) are visible in Datadog + alongside Prometheus. + """ + try: + from litellm.litellm_core_utils.litellm_logging import ( + StandardLoggingPayloadSetup, + ) + from litellm.litellm_core_utils.safe_json_dumps import safe_dumps + + error_information = StandardLoggingPayloadSetup.get_error_information( + original_exception=original_exception, + traceback_str=traceback_str, + ) + _code = error_information.get("error_code") or "" + status_code: Optional[int] = None + if _code and str(_code).strip().isdigit(): + status_code = int(_code) + + # Use project-standard sanitized user context when running in proxy + user_context: Dict[str, Any] = {} + try: + from litellm.proxy.litellm_pre_call_utils import ( + LiteLLMProxyRequestSetup, + ) + + _meta = ( + LiteLLMProxyRequestSetup.get_sanitized_user_information_from_key( + user_api_key_dict=user_api_key_dict + ) + ) + user_context = dict(_meta) if isinstance(_meta, dict) else _meta + except Exception: + # Fallback if proxy not available (e.g. SDK-only): minimal safe fields + if hasattr(user_api_key_dict, "request_route"): + user_context["request_route"] = getattr( + user_api_key_dict, "request_route", None + ) + if hasattr(user_api_key_dict, "team_id"): + user_context["team_id"] = getattr( + user_api_key_dict, "team_id", None + ) + if hasattr(user_api_key_dict, "user_id"): + user_context["user_id"] = getattr( + user_api_key_dict, "user_id", None + ) + if hasattr(user_api_key_dict, "end_user_id"): + user_context["end_user_id"] = getattr( + user_api_key_dict, "end_user_id", None + ) + + message_payload: DatadogProxyFailureHookJsonMessage = { + "exception": error_information.get("error_message") + or str(original_exception), + "error_class": error_information.get("error_class") + or original_exception.__class__.__name__, + "status_code": status_code, + "traceback": error_information.get("traceback") or "", + "user_api_key_dict": user_context, + } + + dd_payload = DatadogPayload( + ddsource=get_datadog_source(), + ddtags=get_datadog_tags(), + hostname=get_datadog_hostname(), + message=safe_dumps(message_payload), + service=get_datadog_service(), + status=DataDogStatus.ERROR, + ) + self._add_trace_context_to_payload(dd_payload=dd_payload) + self.log_queue.append(dd_payload) + + if len(self.log_queue) >= self.batch_size: + await self.async_send_batch() + except Exception as e: + verbose_logger.exception( + f"Datadog: async_post_call_failure_hook - {str(e)}\n{traceback.format_exc()}" + ) + return None + async def async_send_batch(self): """ Sends the in memory logs queue to datadog api @@ -230,9 +329,11 @@ class DataDogLogger( len(self.log_queue), self.intake_url, ) - + if self.is_mock_mode: - verbose_logger.debug("[DATADOG MOCK] Mock mode enabled - API calls will be intercepted") + verbose_logger.debug( + "[DATADOG MOCK] Mock mode enabled - API calls will be intercepted" + ) response = await self.async_send_compressed_data(self.log_queue) if response.status_code == 413: From c9df996b7725b17f095ffced523ca2234b51de31 Mon Sep 17 00:00:00 2001 From: jwang-gif Date: Sat, 7 Feb 2026 22:44:17 -0800 Subject: [PATCH 13/50] Add team policy mapping for zguard (#20608) * support policy mapping on team key level * update document * update document * address comments * update document * add unit test for new feature * add more test case --- .../docs/proxy/guardrails/zscaler_ai_guard.md | 28 +++- .../zscaler_ai_guard/zscaler_ai_guard.py | 60 +++++--- .../guardrails_tests/test_zscaler_ai_guard.py | 129 +++++++++++++++++- 3 files changed, 196 insertions(+), 21 deletions(-) diff --git a/docs/my-website/docs/proxy/guardrails/zscaler_ai_guard.md b/docs/my-website/docs/proxy/guardrails/zscaler_ai_guard.md index 94f31c3bfdf..2e626004238 100644 --- a/docs/my-website/docs/proxy/guardrails/zscaler_ai_guard.md +++ b/docs/my-website/docs/proxy/guardrails/zscaler_ai_guard.md @@ -100,7 +100,7 @@ In cases where encounter other errors when apply Zscaler AI Guard, return exampl } } ``` -## 6. Sending User Information to Zscaler AI Guard for Analysis (Optional) +## 6. Sending User Information to Zscaler AI Guard (Optional) If you need to send end-user information to Zscaler AI Guard for analysis, you can set the configuration in the environment variables to True and include the relevant information in custom_headers on Zscaler AI Guard. - To send user_api_key_alias: @@ -133,4 +133,30 @@ curl -i http://localhost:8165/v1/chat/completions \ "zguard_policy_id": } }' +``` + +## 8. Set Custom Zscaler AI Guard Policy on Litellm Team OR Key Metadata (Optional) +In addition to setting `zguard_policy_id` in a request or the configuration file, you can also set it in the metadata for LiteLLM Team or Key. The `zguard_policy_id` is determined using the following order of precedence: request, Key, Team, config file. This logic is illustrated below: +``` +user_api_key_metadata = metadata.get("user_api_key_metadata", {}) or {} +team_metadata = metadata.get("team_metadata", {}) or {} +policy_id = ( + metadata.get("zguard_policy_id") + if "zguard_policy_id" in metadata + else ( + user_api_key_metadata.get("zguard_policy_id") + if "zguard_policy_id" in user_api_key_metadata + else ( + team_metadata.get("zguard_policy_id") + if "zguard_policy_id" in team_metadata + else self.policy_id + ) + ) + ) +``` +You can leverage this feature to apply multiple policies configured on the Zscaler AI Guard (ZGuard) to traffic from different applications. (Note: It is recommended to map policies using either Team or Key metadata, but not a mix of both.) + +Example set in Team/Key Metadata, you can set From UI: +``` +{"zguard_policy_id": 100} ``` \ No newline at end of file diff --git a/litellm/proxy/guardrails/guardrail_hooks/zscaler_ai_guard/zscaler_ai_guard.py b/litellm/proxy/guardrails/guardrail_hooks/zscaler_ai_guard/zscaler_ai_guard.py index d62bbb0b459..c60752d7952 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/zscaler_ai_guard/zscaler_ai_guard.py +++ b/litellm/proxy/guardrails/guardrail_hooks/zscaler_ai_guard/zscaler_ai_guard.py @@ -92,14 +92,34 @@ class ZscalerAIGuard(CustomGuardrail): Raises: Exception: If content is blocked by Zscaler AI Guard """ + texts = inputs.get("texts", []) try: verbose_proxy_logger.debug(f"ZscalerAIGuard: Checking {len(texts)} text(s)") + metadata = request_data.get("metadata", {}) - custom_policy_id = request_data.get("metadata", {}).get( - "zguard_policy_id", self.policy_id + user_api_key_metadata = metadata.get("user_api_key_metadata", {}) or {} + team_metadata = metadata.get("team_metadata", {}) or {} + + # Precedence for policy_id: + # 1. metadata.zguard_policy_id # request level + # 2. user_api_key_metadata.zguard_policy_id # Key level + # 3. team_metadata.zguard_policy_id # Team level + # 4. self.policy_id (from environment) # Global + policy_id = ( + metadata.get("zguard_policy_id") + if "zguard_policy_id" in metadata + else ( + user_api_key_metadata.get("zguard_policy_id") + if "zguard_policy_id" in user_api_key_metadata + else ( + team_metadata.get("zguard_policy_id") + if "zguard_policy_id" in team_metadata + else self.policy_id + ) + ) ) - verbose_proxy_logger.debug(f"custom_policy_id: {custom_policy_id}") + verbose_proxy_logger.info(f"policy_id applied: {policy_id}") kwargs = {} if self.send_user_api_key_alias: @@ -116,27 +136,29 @@ class ZscalerAIGuard(CustomGuardrail): ) verbose_proxy_logger.debug(f"inside apply_guardrail kwargs: {kwargs}") - # Check each text (Zscaler processes one at a time) - for text in texts: + zscaler_ai_guard_result = None + direction = "OUT" if input_type == "response" else "IN" + verbose_proxy_logger.debug(f"direction: {direction}") + # Concatenate all texts and send to Zscaler AI Guard + if texts: + concatenated_text = " ".join(texts) zscaler_ai_guard_result = await self.make_zscaler_ai_guard_api_call( zscaler_ai_guard_url=self.zscaler_ai_guard_url, api_key=self.api_key, - policy_id=self.policy_id, - direction="IN", - content=text, + policy_id=policy_id, + direction=direction, + content=concatenated_text, **kwargs, ) - - if ( - zscaler_ai_guard_result - and zscaler_ai_guard_result.get("action") == "BLOCK" - ): - blocking_info = zscaler_ai_guard_result.get( - "zscaler_ai_guard_response" - ) - error_message = f"Content blocked by Zscaler AI Guard: {self.extract_blocking_info(blocking_info)}" - raise Exception(error_message) - + if ( + zscaler_ai_guard_result + and zscaler_ai_guard_result.get("action") == "BLOCK" + ): + blocking_info = zscaler_ai_guard_result.get( + "zscaler_ai_guard_response" + ) + error_message = f"Content blocked by Zscaler AI Guard: {self.extract_blocking_info(blocking_info)}" + raise Exception(error_message) except Exception as e: verbose_proxy_logger.error( "ZscalerAIGuard: Failed to apply guardrail: %s", str(e) diff --git a/tests/guardrails_tests/test_zscaler_ai_guard.py b/tests/guardrails_tests/test_zscaler_ai_guard.py index cf70af510c8..9d519c17f6a 100644 --- a/tests/guardrails_tests/test_zscaler_ai_guard.py +++ b/tests/guardrails_tests/test_zscaler_ai_guard.py @@ -116,4 +116,131 @@ def test_extract_blocking_info(): blocking_info = guardrail.extract_blocking_info(response) assert blocking_info["transactionId"] == "12345" - assert blocking_info["blockingDetectors"] == ["detector1"] \ No newline at end of file + assert blocking_info["blockingDetectors"] == ["detector1"] + + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_apply_guardrail_text_concatenation(mock_api_call): + """ + Test that `apply_guardrail` correctly concatenates texts. + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["Hello", "world"]} + request_data = {} + + await guardrail.apply_guardrail(inputs, request_data, "request") + + mock_api_call.assert_called_once() + call_args = mock_api_call.call_args + assert call_args.kwargs["content"] == "Hello world" + + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_policy_id_from_request_metadata(mock_api_call): + """ + Test policy_id is picked from request metadata (highest precedence). + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["test"]} + request_data = { + "metadata": { + "zguard_policy_id": 1, + "user_api_key_metadata": {"zguard_policy_id": 2}, + "team_metadata": {"zguard_policy_id": 3}, + } + } + + await guardrail.apply_guardrail(inputs, request_data, "request") + + mock_api_call.assert_called_once() + assert mock_api_call.call_args.kwargs["policy_id"] == 1 + + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_policy_id_from_user_api_key_metadata(mock_api_call): + """ + Test policy_id is picked from user_api_key_metadata (2nd precedence). + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["test"]} + request_data = { + "metadata": { + "user_api_key_metadata": {"zguard_policy_id": 2}, + "team_metadata": {"zguard_policy_id": 3}, + } + } + + await guardrail.apply_guardrail(inputs, request_data, "request") + + mock_api_call.assert_called_once() + assert mock_api_call.call_args.kwargs["policy_id"] == 2 + + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_policy_id_from_team_metadata(mock_api_call): + """ + Test policy_id is picked from team_metadata (3rd precedence). + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["test"]} + request_data = {"metadata": {"team_metadata": {"zguard_policy_id": 3}}} + + await guardrail.apply_guardrail(inputs, request_data, "request") + + mock_api_call.assert_called_once() + assert mock_api_call.call_args.kwargs["policy_id"] == 3 + + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_policy_id_from_init(mock_api_call): + """ + Test policy_id is picked from guardrail initialization (lowest precedence). + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["test"]} + request_data = {"metadata": {}} + + await guardrail.apply_guardrail(inputs, request_data, "request") + + mock_api_call.assert_called_once() + assert mock_api_call.call_args.kwargs["policy_id"] == 100 + +@pytest.mark.asyncio +@patch( + "litellm.proxy.guardrails.guardrail_hooks.zscaler_ai_guard.ZscalerAIGuard.make_zscaler_ai_guard_api_call", + new_callable=AsyncMock, +) +async def test_policy_id_zero_from_request_metadata(mock_api_call): + """ + Test policy_id=0 is correctly picked. Make sure pick exact policy_id which users set + """ + guardrail = ZscalerAIGuard(policy_id=100) + inputs = {"texts": ["test"]} + request_data = { + "metadata": { + "zguard_policy_id": 0, + } + } + await guardrail.apply_guardrail(inputs, request_data, "request") + mock_api_call.assert_called_once() + assert mock_api_call.call_args.kwargs["policy_id"] == 0 From 55a89f279f219fc9b5a9527581a3b449c0fb8e4c Mon Sep 17 00:00:00 2001 From: nuernber Date: Sat, 7 Feb 2026 22:51:06 -0800 Subject: [PATCH 14/50] feat: add support for anthropic_messages call type in prompt caching (#19233) * feat: add support for anthropic_messages call type in prompt caching * test: move anthropic_messages prompt caching test to main router test file * add tutorial on using claude code with prompt cache routing --- .../claude_code_prompt_cache_routing.md | 43 +++++++ docs/my-website/sidebars.js | 1 + .../prompt_caching_deployment_check.py | 3 +- tests/test_litellm/test_router.py | 121 ++++++++++++++++++ 4 files changed, 167 insertions(+), 1 deletion(-) create mode 100644 docs/my-website/docs/tutorials/claude_code_prompt_cache_routing.md diff --git a/docs/my-website/docs/tutorials/claude_code_prompt_cache_routing.md b/docs/my-website/docs/tutorials/claude_code_prompt_cache_routing.md new file mode 100644 index 00000000000..bbb29489856 --- /dev/null +++ b/docs/my-website/docs/tutorials/claude_code_prompt_cache_routing.md @@ -0,0 +1,43 @@ +# Claude Code - Prompt Cache Routing + +Claude's [Prompt Caching](https://platform.claude.com/docs/en/build-with-claude/prompt-caching) feature helps to optimize API usage through attempting to cache prompts and re-use cached prompts during subsequent API calls. This feature is used by Claude Code. + +When LiteLLM [load balancing](../proxy/load_balancing.md) is enabled, to ensure this prompt caching feature still works with Claude Code, LiteLLM needs to be configured to use the `PromptCachingDeploymentCheck` pre-call check. This pre-call check will ensure that API calls that used prompt caching are remembered and that subsequent API calls that try to use that prompt caching are routed to the same model deployment where a cache write occurred. + +## Set Up + +1. Configure the router so that it uses the `PromptCachingDeploymentCheck` (via setting the `optional_pre_call_checks` property), and configure the models so that they can access multiple deployments of Claude; below, we show an example for multiple AWS accounts (referred to as `account-1` and `account-2`, using the `aws_profile_name` property): +```yaml +router_settings: + optional_pre_call_checks: ["prompt_caching"] + +model_list: +- litellm_params: + model: us.anthropic.claude-sonnet-4-5-20250929-v1:0 + aws_profile_name: account-1 + aws_region_name: us-west-2 + model_info: + litellm_provider: bedrock + model_name: us.anthropic.claude-sonnet-4-5-20250929-v1:0 +- litellm_params: + model: us.anthropic.claude-sonnet-4-5-20250929-v1:0 + aws_profile_name: account-2 + aws_region_name: us-west-2 + model_info: + litellm_provider: bedrock + model_name: us.anthropic.claude-sonnet-4-5-20250929-v1:0 +``` +2. Utilize Claude Code: + 1. Launch Claude Code, which will do a warm-up API call that tries to cache its warm-up prompt and its system prompt. + 2. Wait a few seconds, then quit Claude Code and re-open it. + 3. You'll notice that the warm-up API call successfully gets a cache hit (if using Claude Code in an IDE like VS Code, ensure that you don't do anything between step 2.1 and 2.2 here, otherwise there may not be a cache hit): + 1. Go to the [LiteLLM Request Logs page](../proxy/ui_logs.md) in the Admin UI + 2. Click on the individual requests to see (a) the cache creation and cache read tokens; and (b) the Model ID. In particular, the API call from step 2.1 should show a cache write, and the API call from step 2.2 should show a cache read; in addition, the Model ID should be equal (meaning the API call is getting forwarded to the same AWS account). + +## Related + +- [Claude Code - Quickstart](./claude_responses_api.md) +- [Claude Code - Customer Tracking](./claude_code_customer_tracking.md) +- [Claude Code - Plugin Marketplace](./claude_code_plugin_marketplace.md) +- [Claude Code - WebSearch](./claude_code_websearch.md) +- [Proxy - Load Balancing](../proxy/load_balancing.md) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 6c354b7c041..9d9007a916c 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -125,6 +125,7 @@ const sidebars = { "tutorials/claude_responses_api", "tutorials/claude_code_max_subscription", "tutorials/claude_code_customer_tracking", + "tutorials/claude_code_prompt_cache_routing", "tutorials/claude_code_websearch", "tutorials/claude_mcp", "tutorials/claude_non_anthropic_models", diff --git a/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py b/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py index d3d237d9f28..e9c4b69d8ef 100644 --- a/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py +++ b/litellm/router_utils/pre_call_checks/prompt_caching_deployment_check.py @@ -61,9 +61,10 @@ class PromptCachingDeploymentCheck(CustomLogger): if ( call_type != CallTypes.completion.value and call_type != CallTypes.acompletion.value + and call_type != CallTypes.anthropic_messages.value ): # only use prompt caching for completion calls verbose_logger.debug( - "litellm.router_utils.pre_call_checks.prompt_caching_deployment_check: skipping adding model id to prompt caching cache, CALL TYPE IS NOT COMPLETION" + "litellm.router_utils.pre_call_checks.prompt_caching_deployment_check: skipping adding model id to prompt caching cache, CALL TYPE IS NOT COMPLETION or ANTHROPIC MESSAGE" ) return diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index 08ae804ea80..75ec806ee17 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -1869,3 +1869,124 @@ async def test_aguardrail(): assert result["result"] == "success" assert result["selected_guardrail"]["id"] == "guardrail-1" + +@pytest.mark.asyncio +async def test_anthropic_messages_call_type_is_cached(): + """ + Regression test: Verify that anthropic_messages call type is allowed + in PromptCachingDeploymentCheck.async_log_success_event. + """ + import asyncio + from litellm.router_utils.pre_call_checks.prompt_caching_deployment_check import ( + PromptCachingDeploymentCheck, + ) + from litellm.router_utils.prompt_caching_cache import PromptCachingCache + from litellm.caching.dual_cache import DualCache + from litellm.types.utils import CallTypes + from litellm.types.utils import ( + StandardLoggingPayload, + StandardLoggingModelInformation, + StandardLoggingMetadata, + StandardLoggingHiddenParams, + ) + + # Create mock standard logging payload inline + def create_standard_logging_payload() -> StandardLoggingPayload: + return StandardLoggingPayload( + id="test_id", + call_type="completion", + response_cost=0.1, + response_cost_failure_debug_info=None, + status="success", + total_tokens=30, + prompt_tokens=20, + completion_tokens=10, + startTime=1234567890.0, + endTime=1234567891.0, + completionStartTime=1234567890.5, + model_map_information=StandardLoggingModelInformation( + model_map_key="gpt-3.5-turbo", model_map_value=None + ), + model="gpt-3.5-turbo", + model_id="model-123", + model_group="openai-gpt", + api_base="https://api.openai.com", + metadata=StandardLoggingMetadata( + user_api_key_hash="test_hash", + user_api_key_org_id=None, + user_api_key_alias="test_alias", + user_api_key_team_id="test_team", + user_api_key_user_id="test_user", + user_api_key_team_alias="test_team_alias", + spend_logs_metadata=None, + requester_ip_address="127.0.0.1", + requester_metadata=None, + ), + cache_hit=False, + cache_key=None, + saved_cache_cost=0.0, + request_tags=[], + end_user=None, + requester_ip_address="127.0.0.1", + messages=[{"role": "user", "content": "Hello, world!"}], + response={"choices": [{"message": {"content": "Hi there!"}}]}, + error_str=None, + model_parameters={"stream": True}, + hidden_params=StandardLoggingHiddenParams( + model_id="model-123", + cache_key=None, + api_base="https://api.openai.com", + response_cost="0.1", + additional_headers=None, + ), + ) + + cache = DualCache() + deployment_check = PromptCachingDeploymentCheck(cache=cache) + prompt_cache = PromptCachingCache(cache=cache) + + # Create messages with enough tokens to pass the caching threshold + test_messages = [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "test long message here" * 1024, + "cache_control": { + "type": "ephemeral", + "ttl": "5m" + } + } + ] + } + ] + test_model_id = "test-model-id-123" + + # Create a payload with anthropic_messages call type + payload = create_standard_logging_payload() + payload["call_type"] = CallTypes.anthropic_messages.value + payload["messages"] = test_messages + payload["model"] = "anthropic/claude-3-5-sonnet-20240620" + payload["model_id"] = test_model_id + + # Log the success event (should cache the model_id) + await deployment_check.async_log_success_event( + kwargs={"standard_logging_object": payload}, + response_obj={}, + start_time=1234567890.0, + end_time=1234567891.0, + ) + + # Small delay to ensure cache write completes + await asyncio.sleep(0.1) + + # Verify that the model_id was actually cached + cached_result = await prompt_cache.async_get_model_id( + messages=test_messages, + tools=None, + ) + + # This assertion will FAIL if anthropic_messages is filtered out + assert cached_result is not None, "Model ID should be cached for anthropic_messages call type" + assert cached_result["model_id"] == test_model_id, f"Expected {test_model_id}, got {cached_result['model_id']}" From 1fecae0399a284cd0c31322cbfdb9a1372f59958 Mon Sep 17 00:00:00 2001 From: Cesar Garcia <128240629+Chesars@users.noreply.github.com> Date: Sun, 8 Feb 2026 03:57:04 -0300 Subject: [PATCH 15/50] docs: add SDK proxy authentication (OAuth2/JWT auto-refresh) documentation (#20680) Adds documentation for the litellm.proxy_auth feature that automatically obtains and refreshes OAuth2/JWT tokens when connecting to a LiteLLM Proxy. --- .../docs/providers/litellm_proxy.md | 22 ++ docs/my-website/docs/proxy_auth.md | 333 ++++++++++++++++++ docs/my-website/sidebars.js | 1 + 3 files changed, 356 insertions(+) create mode 100644 docs/my-website/docs/proxy_auth.md diff --git a/docs/my-website/docs/providers/litellm_proxy.md b/docs/my-website/docs/providers/litellm_proxy.md index bfefc8a787c..918ac6755a5 100644 --- a/docs/my-website/docs/providers/litellm_proxy.md +++ b/docs/my-website/docs/providers/litellm_proxy.md @@ -227,6 +227,28 @@ response = litellm.completion( ) ``` +## OAuth2/JWT Authentication + +If your LiteLLM Proxy requires OAuth2/JWT authentication (e.g., Azure AD, Keycloak, Okta), the SDK can automatically obtain and refresh tokens for you. + +```python +import litellm +from litellm.proxy_auth import AzureADCredential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=AzureADCredential(), + scope="api://my-litellm-proxy/.default" +) +litellm.api_base = "https://my-proxy.example.com" + +response = litellm.completion( + model="gpt-4", + messages=[{"role": "user", "content": "Hello!"}] +) +``` + +[Learn more about SDK Proxy Authentication (OAuth2/JWT Auto-Refresh) β†’](../proxy_auth) + ## Sending `tags` to LiteLLM Proxy Tags allow you to categorize and track your API requests for monitoring, debugging, and analytics purposes. You can send tags as a list of strings to the LiteLLM Proxy using the `extra_body` parameter. diff --git a/docs/my-website/docs/proxy_auth.md b/docs/my-website/docs/proxy_auth.md new file mode 100644 index 00000000000..91084b34a37 --- /dev/null +++ b/docs/my-website/docs/proxy_auth.md @@ -0,0 +1,333 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# SDK Proxy Authentication (OAuth2/JWT Auto-Refresh) + +Automatically obtain and refresh OAuth2/JWT tokens when using the LiteLLM Python SDK with a LiteLLM Proxy that requires JWT authentication. + +## Overview + +When your LiteLLM Proxy is protected by an OAuth2/OIDC provider (Azure AD, Keycloak, Okta, Auth0, etc.), your SDK clients need valid JWT tokens for every request. Instead of manually managing token lifecycle, `litellm.proxy_auth` handles this automatically: + +- Obtains tokens from your identity provider +- Caches tokens to avoid unnecessary requests +- Refreshes tokens before they expire (60-second buffer) +- Injects `Authorization: Bearer ` headers into every request + +## Quick Start + +### Azure AD + + + + +Uses the [DefaultAzureCredential](https://learn.microsoft.com/en-us/python/api/azure-identity/azure.identity.defaultazurecredential) chain (environment variables, managed identity, Azure CLI, etc.): + +```python +import litellm +from litellm.proxy_auth import AzureADCredential, ProxyAuthHandler + +# One-time setup +litellm.proxy_auth = ProxyAuthHandler( + credential=AzureADCredential(), # uses DefaultAzureCredential + scope="api://my-litellm-proxy/.default" +) +litellm.api_base = "https://my-proxy.example.com" + +# All requests now include Authorization headers automatically +response = litellm.completion( + model="gpt-4", + messages=[{"role": "user", "content": "Hello!"}] +) +``` + + + + +Use a specific Azure AD app registration: + +```python +import litellm +from azure.identity import ClientSecretCredential +from litellm.proxy_auth import AzureADCredential, ProxyAuthHandler + +azure_cred = ClientSecretCredential( + tenant_id="your-tenant-id", + client_id="your-client-id", + client_secret="your-client-secret" +) + +litellm.proxy_auth = ProxyAuthHandler( + credential=AzureADCredential(credential=azure_cred), + scope="api://my-litellm-proxy/.default" +) +litellm.api_base = "https://my-proxy.example.com" + +response = litellm.completion( + model="gpt-4", + messages=[{"role": "user", "content": "Hello!"}] +) +``` + + + + +**Required package:** `pip install azure-identity` + +### Generic OAuth2 (Okta, Auth0, Keycloak, etc.) + +Works with any OAuth2 provider that supports the `client_credentials` grant type: + +```python +import litellm +from litellm.proxy_auth import GenericOAuth2Credential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=GenericOAuth2Credential( + client_id="your-client-id", + client_secret="your-client-secret", + token_url="https://your-idp.example.com/oauth2/token" + ), + scope="litellm_proxy_api" +) +litellm.api_base = "https://my-proxy.example.com" + +response = litellm.completion( + model="gpt-4", + messages=[{"role": "user", "content": "Hello!"}] +) +``` + +### Custom Credential Provider + +Implement the `TokenCredential` protocol to use any authentication mechanism: + +```python +import time +import litellm +from litellm.proxy_auth import AccessToken, ProxyAuthHandler + +class MyCustomCredential: + """Any class with a get_token(scope) -> AccessToken method works.""" + + def get_token(self, scope: str) -> AccessToken: + # Your custom logic to obtain a token + token = my_auth_system.get_jwt(scope=scope) + return AccessToken( + token=token, + expires_on=int(time.time()) + 3600 + ) + +litellm.proxy_auth = ProxyAuthHandler( + credential=MyCustomCredential(), + scope="my-scope" +) +``` + +## Supported Endpoints + +Auth headers are automatically injected for: + +| Endpoint | Function | +|----------|----------| +| Chat Completions | `litellm.completion()` / `litellm.acompletion()` | +| Embeddings | `litellm.embedding()` / `litellm.aembedding()` | + +## How It Works + +``` +β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” +β”‚ Your β”‚ β”‚ ProxyAuthHandler β”‚ β”‚ Identity β”‚ β”‚ LiteLLM β”‚ +β”‚ Code │────▢│ (token cache) │────▢│ Provider β”‚ β”‚ Proxy β”‚ +β”‚ β”‚ β”‚ │◀────│ (Azure AD, β”‚ β”‚ β”‚ +β”‚ β”‚ β”‚ β”‚ β”‚ Okta, etc) β”‚ β”‚ β”‚ +β”‚ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β”‚ +β”‚ β”‚ β”‚ Authorization: Bearer β”‚ β”‚ +β”‚ │──────────────┼───────────────────────────────────▢│ β”‚ +β”‚ │◀─────────────┼────────────────────────────────────│ β”‚ +β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ +``` + +1. You set `litellm.proxy_auth` once at startup +2. On each SDK call (`completion()`, `embedding()`), the handler checks its cached token +3. If the token is missing or expires within 60 seconds, it requests a new one from your identity provider +4. The `Authorization: Bearer ` header is injected into the request +5. If token retrieval fails, a warning is logged and the request proceeds without auth headers + +## API Reference + +### ProxyAuthHandler + +The main handler that manages the token lifecycle. + +```python +from litellm.proxy_auth import ProxyAuthHandler + +handler = ProxyAuthHandler( + credential=, # required - credential provider + scope="" # required - OAuth2 scope to request +) +``` + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `credential` | `TokenCredential` | Yes | A credential provider (AzureADCredential, GenericOAuth2Credential, or custom) | +| `scope` | `str` | Yes | The OAuth2 scope to request tokens for | + +**Methods:** + +| Method | Returns | Description | +|--------|---------|-------------| +| `get_token()` | `AccessToken` | Get a valid token, refreshing if needed | +| `get_auth_headers()` | `dict` | Get `{"Authorization": "Bearer "}` headers | + +### AzureADCredential + +Wraps any `azure-identity` credential with lazy initialization. + +```python +from litellm.proxy_auth import AzureADCredential + +# Uses DefaultAzureCredential (recommended) +cred = AzureADCredential() + +# Or wrap a specific azure-identity credential +from azure.identity import ManagedIdentityCredential +cred = AzureADCredential(credential=ManagedIdentityCredential()) +``` + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `credential` | Azure `TokenCredential` | No | An azure-identity credential. If `None`, uses `DefaultAzureCredential` | + +### GenericOAuth2Credential + +Standard OAuth2 client credentials flow for any provider. + +```python +from litellm.proxy_auth import GenericOAuth2Credential + +cred = GenericOAuth2Credential( + client_id="your-client-id", + client_secret="your-client-secret", + token_url="https://your-idp.com/oauth2/token" +) +``` + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `client_id` | `str` | Yes | OAuth2 client ID | +| `client_secret` | `str` | Yes | OAuth2 client secret | +| `token_url` | `str` | Yes | Token endpoint URL | + +### AccessToken + +Dataclass representing an OAuth2 access token. + +```python +from litellm.proxy_auth import AccessToken + +token = AccessToken( + token="eyJhbG...", # JWT string + expires_on=1234567890 # Unix timestamp +) +``` + +### TokenCredential Protocol + +Any class implementing this protocol can be used as a credential provider: + +```python +from litellm.proxy_auth import AccessToken + +class MyCredential: + def get_token(self, scope: str) -> AccessToken: + ... +``` + +## Provider-Specific Examples + +### Keycloak + +```python +from litellm.proxy_auth import GenericOAuth2Credential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=GenericOAuth2Credential( + client_id="litellm-client", + client_secret="your-keycloak-client-secret", + token_url="https://keycloak.example.com/realms/your-realm/protocol/openid-connect/token" + ), + scope="openid" +) +``` + +### Okta + +```python +from litellm.proxy_auth import GenericOAuth2Credential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=GenericOAuth2Credential( + client_id="your-okta-client-id", + client_secret="your-okta-client-secret", + token_url="https://your-org.okta.com/oauth2/default/v1/token" + ), + scope="litellm_api" +) +``` + +### Auth0 + +```python +from litellm.proxy_auth import GenericOAuth2Credential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=GenericOAuth2Credential( + client_id="your-auth0-client-id", + client_secret="your-auth0-client-secret", + token_url="https://your-tenant.auth0.com/oauth/token" + ), + scope="https://my-proxy.example.com/api" +) +``` + +### Azure AD with Managed Identity + +```python +from azure.identity import ManagedIdentityCredential +from litellm.proxy_auth import AzureADCredential, ProxyAuthHandler + +litellm.proxy_auth = ProxyAuthHandler( + credential=AzureADCredential( + credential=ManagedIdentityCredential() + ), + scope="api://my-litellm-proxy/.default" +) +``` + +## Combining with `use_litellm_proxy` + +You can use `proxy_auth` together with [`use_litellm_proxy`](./providers/litellm_proxy#send-all-sdk-requests-to-litellm-proxy) to route all SDK requests through an authenticated proxy: + +```python +import os +import litellm +from litellm.proxy_auth import AzureADCredential, ProxyAuthHandler + +# Route all requests through the proxy +os.environ["LITELLM_PROXY_API_BASE"] = "https://my-proxy.example.com" +litellm.use_litellm_proxy = True + +# Authenticate with OAuth2/JWT +litellm.proxy_auth = ProxyAuthHandler( + credential=AzureADCredential(), + scope="api://my-litellm-proxy/.default" +) + +# This request goes through the proxy with automatic JWT auth +response = litellm.completion( + model="vertex_ai/gemini-2.0-flash-001", + messages=[{"role": "user", "content": "Hello!"}] +) +``` diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 9d9007a916c..343860cb158 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -224,6 +224,7 @@ const sidebars = { label: "Configuration", items: [ "set_keys", + "proxy_auth", "caching/all_caches", ], }, From 8dcd18301366b69d3cc808acfe011922e8d5e24b Mon Sep 17 00:00:00 2001 From: John Lathouwers Date: Sun, 8 Feb 2026 06:58:59 +0000 Subject: [PATCH 16/50] Fixes #20582 (#20663) --- litellm/llms/oci/chat/transformation.py | 37 +++++- litellm/types/llms/oci.py | 58 +++++----- .../oci/chat/test_oci_chat_transformation.py | 108 ++++++++++++++++++ .../oci/chat/test_oci_cohere_tool_calls.py | 104 +++++++++++++++++ 4 files changed, 280 insertions(+), 27 deletions(-) diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index 84f39ef2525..e66394ae5f5 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -218,6 +218,7 @@ class OCIChatConfig(BaseConfig): "parallel_tool_calls": False, "audio": False, "web_search_options": False, + "response_format": "responseFormat", } # Cohere and Gemini use the same parameter mapping as GENERIC @@ -269,6 +270,9 @@ class OCIChatConfig(BaseConfig): adapted_params[alias] = value + if alias == "responseFormat": + adapted_params["response_format"] = value + return adapted_params def _sign_with_oci_signer( @@ -673,6 +677,36 @@ class OCIChatConfig(BaseConfig): selected_params["tools"] = adapt_tool_definition_to_oci_standard( # type: ignore[assignment] selected_params["tools"], vendor # type: ignore[arg-type] ) + + # Transform response_format type to OCI uppercase format + if "responseFormat" in selected_params: + rf = selected_params["responseFormat"] + if isinstance(rf, dict) and "type" in rf: + rf_payload = dict(rf) + selected_params["responseFormat"] = rf_payload + + response_type = rf_payload["type"] + schema_payload: Optional[Any] = None + + if "json_schema" in rf_payload: + raw_schema_payload = rf_payload.pop("json_schema") + if isinstance(raw_schema_payload, dict): + schema_payload = dict(raw_schema_payload) + else: + schema_payload = raw_schema_payload + + if schema_payload is not None: + rf_payload["jsonSchema"] = schema_payload + + if vendor == OCIVendors.COHERE: + # Cohere expects lower-case type values + rf_payload["type"] = response_type + else: + format_type = response_type.upper() + if format_type == "JSON": + format_type = "JSON_OBJECT" + rf_payload["type"] = format_type + return selected_params def adapt_messages_to_cohere_standard(self, messages: List[AllMessageValues]) -> List[CohereMessage]: @@ -806,11 +840,12 @@ class OCIChatConfig(BaseConfig): # Create Cohere-specific chat request + optional_cohere_params = self._get_optional_params(OCIVendors.COHERE, optional_params) chat_request = CohereChatRequest( apiFormat="COHERE", message=self._extract_text_content(user_messages[-1]["content"]), chatHistory=self.adapt_messages_to_cohere_standard(messages), - **self._get_optional_params(OCIVendors.COHERE, optional_params) + **optional_cohere_params ) data = OCICompletionPayload( diff --git a/litellm/types/llms/oci.py b/litellm/types/llms/oci.py index 9a654bc0f6c..cb1dd391434 100644 --- a/litellm/types/llms/oci.py +++ b/litellm/types/llms/oci.py @@ -102,6 +102,7 @@ class OCIChatRequestPayload(BaseModel): seed: Optional[int] = None frequencyPenalty: Optional[float] = None presencePenalty: Optional[float] = None + responseFormat: Optional[Dict[str, Any]] = None class OCIServingMode(BaseModel): @@ -125,14 +126,14 @@ class OCICompletionPayload(BaseModel): class OCICompletionTokenDetails(BaseModel): """Completion token details in the OCI response.""" - acceptedPredictionTokens: int - reasoningTokens: int + acceptedPredictionTokens: Optional[int] = None + reasoningTokens: Optional[int] = None class OCIPromptTokensDetails(BaseModel): """Prompt token details in the OCI response.""" - cachedTokens: int + cachedTokens: Optional[int] = None class OCIResponseUsage(BaseModel): @@ -205,40 +206,40 @@ class CohereStreamChunk(BaseModel): class CohereMessage(BaseModel): """Base model for Cohere messages.""" - + role: str - message: str + message: Optional[str] = None toolCalls: Optional[List[CohereToolCall]] = None class CohereUserMessage(CohereMessage): """User message in Cohere chat.""" - + role: Literal["USER"] = "USER" class CohereChatBotMessage(CohereMessage): """Chatbot message in Cohere chat.""" - + role: Literal["CHATBOT"] = "CHATBOT" class CohereSystemMessage(CohereMessage): """System message in Cohere chat.""" - + role: Literal["SYSTEM"] = "SYSTEM" class CohereToolMessage(CohereMessage): """Tool message in Cohere chat.""" - + role: Literal["TOOL"] = "TOOL" toolCallId: str class CohereParameterDefinition(BaseModel): """Parameter definition for Cohere tools.""" - + description: str type: str isRequired: bool = False @@ -246,7 +247,7 @@ class CohereParameterDefinition(BaseModel): class CohereTool(BaseModel): """Tool definition for Cohere.""" - + name: str description: str parameterDefinitions: Dict[str, CohereParameterDefinition] @@ -254,38 +255,44 @@ class CohereTool(BaseModel): class CohereToolCall(BaseModel): """Tool call made by Cohere model.""" - + name: str parameters: Dict[str, Any] class CohereToolResult(BaseModel): """Result of a tool call.""" - + callId: str result: str class CohereResponseFormat(BaseModel): """Response format for Cohere.""" - + type: str class CohereResponseTextFormat(CohereResponseFormat): """Text response format for Cohere.""" - + type: Literal["text"] = "text" +class CohereResponseJSONSchemaFormat(CohereResponseFormat): + """JSON schema response format for Cohere.""" + + type: Literal["json_schema"] = "json_schema" + jsonSchema: Dict[str, Any] + class CohereChatRequest(BaseModel): """Cohere chat request model.""" - + # Required fields message: str apiFormat: Literal["COHERE"] = "COHERE" - + # Optional fields chatHistory: Optional[List[CohereMessage]] = None maxTokens: Optional[int] = None @@ -298,7 +305,7 @@ class CohereChatRequest(BaseModel): seed: Optional[int] = None tools: Optional[List[CohereTool]] = None toolChoice: Optional[Union[str, Dict[str, Any]]] = None - responseFormat: Optional[CohereResponseFormat] = None + responseFormat: Optional[Union[CohereResponseTextFormat, CohereResponseJSONSchemaFormat, CohereResponseFormat]] = None preambleOverride: Optional[str] = None documents: Optional[List[Dict[str, Any]]] = None searchQueriesOnly: Optional[bool] = None @@ -318,7 +325,7 @@ class CohereChatRequest(BaseModel): class CohereUsage(BaseModel): """Usage information for Cohere response.""" - + promptTokens: int completionTokens: int totalTokens: int @@ -328,7 +335,7 @@ class CohereUsage(BaseModel): class CohereCitation(BaseModel): """Citation in Cohere response.""" - + start: int end: int text: str @@ -337,19 +344,19 @@ class CohereCitation(BaseModel): class CohereSearchQuery(BaseModel): """Search query generated by Cohere.""" - + text: str generation_id: str class CohereChatResponse(BaseModel): """Cohere chat response model.""" - + # Required fields text: str apiFormat: Literal["COHERE"] = "COHERE" finishReason: Literal["COMPLETE", "ERROR_TOXIC", "ERROR_LIMIT", "ERROR", "USER_CANCEL", "MAX_TOKENS"] - + # Optional fields chatHistory: Optional[List[CohereMessage]] = None citations: Optional[List[CohereCitation]] = None @@ -364,7 +371,7 @@ class CohereChatResponse(BaseModel): class CohereChatDetails(BaseModel): """Chat details for Cohere request.""" - + compartmentId: str servingMode: OCIServingMode chatRequest: CohereChatRequest @@ -372,8 +379,7 @@ class CohereChatDetails(BaseModel): class CohereChatResult(BaseModel): """Complete Cohere chat result.""" - + modelId: str modelVersion: str chatResponse: CohereChatResponse - diff --git a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py index 3bd46b84e6c..3b53f9de714 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_chat_transformation.py @@ -287,6 +287,114 @@ class TestOCIChatConfig: # Verify the message content assert transformed_request["chatRequest"]["message"] == "What is quantum computing?" + def test_transform_request_response_format_json_object(self): + """ + Tests that response_format type 'json_object' is uppercased to 'JSON_OBJECT' for generic OCI models. + """ + config = OCIChatConfig() + optional_params = { + "oci_compartment_id": TEST_COMPARTMENT_ID, + "response_format": {"type": "json_object"}, + } + transformed_request = config.transform_request( + model=TEST_MODEL_NAME, + messages=TEST_MESSAGES, # type: ignore + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + rf = transformed_request["chatRequest"]["responseFormat"] + assert rf["type"] == "JSON_OBJECT" + + def test_transform_request_response_format_text(self): + """ + Tests that response_format type 'text' is uppercased to 'TEXT' for generic OCI models. + """ + config = OCIChatConfig() + optional_params = { + "oci_compartment_id": TEST_COMPARTMENT_ID, + "response_format": {"type": "text"}, + } + transformed_request = config.transform_request( + model=TEST_MODEL_NAME, + messages=TEST_MESSAGES, # type: ignore + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + rf = transformed_request["chatRequest"]["responseFormat"] + assert rf["type"] == "TEXT" + + def test_transform_request_response_format_json_shorthand(self): + """ + Tests that response_format type 'json' is mapped to 'JSON_OBJECT' for generic OCI models. + """ + config = OCIChatConfig() + optional_params = { + "oci_compartment_id": TEST_COMPARTMENT_ID, + "response_format": {"type": "json"}, + } + transformed_request = config.transform_request( + model=TEST_MODEL_NAME, + messages=TEST_MESSAGES, # type: ignore + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + rf = transformed_request["chatRequest"]["responseFormat"] + assert rf["type"] == "JSON_OBJECT" + + def test_transform_response_without_token_details(self): + """ + Tests that responses missing completionTokensDetails and promptTokensDetails + are handled correctly (fields are optional). + """ + config = OCIChatConfig() + created_time = datetime.datetime.now(datetime.timezone.utc).isoformat().replace("+00:00", "Z") + mock_oci_response = { + "modelId": TEST_MODEL_NAME, + "modelVersion": "1.0", + "chatResponse": { + "apiFormat": "GENERIC", + "choices": [ + { + "index": 0, + "message": { + "role": "ASSISTANT", + "content": [{"type": "TEXT", "text": "Hello!"}], + }, + "finishReason": "STOP", + } + ], + "timeCreated": created_time, + "usage": { + "promptTokens": 5, + "completionTokens": 10, + "totalTokens": 15, + }, + }, + } + response = httpx.Response( + status_code=200, json=mock_oci_response, headers={"Content-Type": "application/json"} + ) + result = config.transform_response( + model=TEST_MODEL_NAME, + raw_response=response, + model_response=ModelResponse(), + logging_obj={}, # type: ignore + request_data={}, + messages=[], + optional_params={}, + litellm_params={}, + encoding={}, + ) + + assert isinstance(result, ModelResponse) + assert result.choices[0].message.content == "Hello!" + assert result.usage.prompt_tokens == 5 # type: ignore + assert result.usage.completion_tokens == 10 # type: ignore + assert result.usage.total_tokens == 15 # type: ignore + def test_transform_response_simple_text(self): """ Tests if a simple text response is transformed correctly. diff --git a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py index abbb7e3e301..a9c4bead820 100644 --- a/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py +++ b/tests/test_litellm/llms/oci/chat/test_oci_cohere_tool_calls.py @@ -239,6 +239,110 @@ class TestOCICohereToolCalls: assert result.usage.completion_tokens == 22 assert result.usage.total_tokens == 48 + def test_cohere_request_preserves_json_schema_response_format(self): + """Ensure Cohere requests retain JSON schema payloads in responseFormat.""" + config = OCIChatConfig() + messages = [{"role": "user", "content": "Return structured info"}] + response_format = { + "type": "json_schema", + "json_schema": { + "name": "test_schema", + "strict": True, + "schema": { + "type": "object", + "properties": { + "foo": {"type": "string"} + }, + "required": ["foo"] + } + } + } + optional_params = { + "oci_compartment_id": TEST_COMPARTMENT_ID, + "response_format": response_format, + } + + transformed_request = config.transform_request( + model="cohere.command-rplus", + messages=messages, # type: ignore[arg-type] + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + chat_request = transformed_request["chatRequest"] + assert chat_request["apiFormat"] == "COHERE" + assert "responseFormat" in chat_request + + cohere_response_format = chat_request["responseFormat"] + assert cohere_response_format["type"] == "json_schema" + assert "json_schema" not in cohere_response_format + assert "jsonSchema" in cohere_response_format + assert cohere_response_format["jsonSchema"] == response_format["json_schema"] + + def test_cohere_request_response_format_text_stays_lowercase(self): + """Ensure Cohere keeps response_format type lowercase (e.g. 'text' not 'TEXT').""" + config = OCIChatConfig() + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "oci_compartment_id": TEST_COMPARTMENT_ID, + "response_format": {"type": "text"}, + } + + transformed_request = config.transform_request( + model="cohere.command-latest", + messages=messages, # type: ignore + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + chat_request = transformed_request["chatRequest"] + assert chat_request["apiFormat"] == "COHERE" + assert "responseFormat" in chat_request + assert chat_request["responseFormat"]["type"] == "text" + + def test_cohere_tool_call_only_message_no_text(self): + """Test chat history with an assistant message that has tool calls but no text content.""" + config = OCIChatConfig() + + messages = [ + {"role": "user", "content": "What's the weather?"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": { + "name": "get_weather", + "arguments": '{"location": "Paris"}', + }, + } + ], + }, + { + "role": "tool", + "content": "Sunny, 25C", + "tool_call_id": "call_1", + }, + ] + + chat_history = config.adapt_messages_to_cohere_standard(messages) + + # First message is the user message + assert chat_history[0].role == "USER" + assert chat_history[0].message == "What's the weather?" + + # Second message is the assistant with tool calls and no text + assistant_msg = chat_history[1] + assert assistant_msg.role == "CHATBOT" + assert assistant_msg.message is None or assistant_msg.message == "" + assert assistant_msg.toolCalls is not None + assert len(assistant_msg.toolCalls) == 1 + assert assistant_msg.toolCalls[0].name == "get_weather" + def test_cohere_chat_history_with_tool_calls(self): """Test chat history transformation with tool calls""" config = OCIChatConfig() From 7335965c12848f5b9cd07c865dc770c29c8281e0 Mon Sep 17 00:00:00 2001 From: Varun Chawla <34209028+veeceey@users.noreply.github.com> Date: Sat, 7 Feb 2026 22:59:49 -0800 Subject: [PATCH 17/50] fix: show error details instead of Data Not Available for failed requests (#20656) --- .../LogDetailsDrawer/LogDetailsDrawer.tsx | 9 ++- .../view_logs/RequestResponsePanel.test.tsx | 74 +++++++++++++++++++ .../view_logs/RequestResponsePanel.tsx | 4 +- .../src/components/view_logs/index.tsx | 2 +- 4 files changed, 83 insertions(+), 6 deletions(-) diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx index 54946eb0964..a3f948296eb 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx @@ -85,7 +85,7 @@ export function LogDetailsDrawer({ // Check if request/response data is present const hasMessages = checkHasMessages(logEntry.messages); const hasResponse = checkHasResponse(logEntry.response); - const missingData = !hasMessages && !hasResponse; + const missingData = !hasMessages && !hasResponse && !hasError; // Guardrail data const guardrailInfo = metadata?.guardrail_information; @@ -206,6 +206,7 @@ export function LogDetailsDrawer({ {/* Request/Response JSON - Collapsible */} any; getFormattedResponse: () => any; logEntry: LogEntry; @@ -346,6 +348,7 @@ interface RequestResponseSectionProps { function RequestResponseSection({ hasResponse, + hasError, getRawRequest, getFormattedResponse, logEntry, @@ -423,7 +426,7 @@ function RequestResponseSection({ text: getCopyText(), tooltips: ["Copy JSON", "Copied!"] }} - disabled={activeTab === TAB_RESPONSE && !hasResponse} + disabled={activeTab === TAB_RESPONSE && !hasResponse && !hasError} /> } items={[ @@ -441,7 +444,7 @@ function RequestResponseSection({ label: "Response", children: (
- {hasResponse ? ( + {hasResponse || hasError ? ( ) : (
diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.test.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.test.tsx index deeac3a8d0b..b7c0318d9fd 100644 --- a/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.test.tsx @@ -188,4 +188,78 @@ describe("RequestResponsePanel", () => { expect(responseData).toEqual({ responseData: "this should appear in response" }); expect(responseData).not.toEqual({ requestData: "this should not appear in response" }); }); + + it("should show error response data when hasError is true and hasResponse is false", () => { + const failedLogEntry: LogEntry = { + ...baseLogEntry, + messages: [], + response: {}, + metadata: { + status: "failure", + error_information: { + error_message: "Model not found", + error_class: "NotFoundError", + error_code: 404, + }, + additional_usage_values: { + cache_read_input_tokens: 0, + cache_creation_input_tokens: 0, + }, + }, + }; + const errorResponse = { error: { message: "Model not found", type: "NotFoundError", code: 404, param: null } }; + const mockGetRawRequest = vi.fn().mockReturnValue({ messages: [] }); + const mockFormattedResponse = vi.fn().mockReturnValue(errorResponse); + render( + , + ); + expect(screen.queryByText("Response data not available")).not.toBeInTheDocument(); + expect(mockFormattedResponse).toHaveBeenCalled(); + const copyButtons = screen.getAllByRole("button"); + const copyResponseButton = copyButtons.find((button) => button.getAttribute("title") === "Copy response"); + expect(copyResponseButton).not.toBeDisabled(); + }); + + it("should show Response data not available when hasResponse and hasError are both false", () => { + const mockGetRawRequest = vi.fn().mockReturnValue({ messages: [] }); + const mockFormattedResponse = vi.fn().mockReturnValue({}); + render( + , + ); + expect(screen.getByText("Response data not available")).toBeInTheDocument(); + }); + + it("should show error code in response header when hasError is true", () => { + const errorInfo = { error_message: "Rate limit exceeded", error_class: "RateLimitError", error_code: 429 }; + const mockGetRawRequest = vi.fn().mockReturnValue({ messages: [] }); + const mockFormattedResponse = vi.fn().mockReturnValue({ error: { message: "Rate limit exceeded", type: "RateLimitError", code: 429, param: null } }); + render( + , + ); + expect(screen.getByText(/HTTP code 429/)).toBeInTheDocument(); + }); }); diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.tsx index b2cae68184a..da9323f8172 100644 --- a/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/RequestResponsePanel.tsx @@ -113,7 +113,7 @@ export function RequestResponsePanel({ onClick={handleCopyResponse} className="p-1 hover:bg-gray-200 rounded" title="Copy response" - disabled={!hasResponse} + disabled={!hasResponse && !hasError} >
- {hasResponse ? ( + {hasResponse || hasError ? (
diff --git a/ui/litellm-dashboard/src/components/view_logs/index.tsx b/ui/litellm-dashboard/src/components/view_logs/index.tsx index 3859a5e51fb..87e11e00c7c 100644 --- a/ui/litellm-dashboard/src/components/view_logs/index.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/index.tsx @@ -822,7 +822,7 @@ export function RequestViewer({ row, onOpenSettings }: { row: Row; onO ? row.original.messages.length > 0 : Object.keys(row.original.messages).length > 0); const hasResponse = row.original.response && Object.keys(formatData(row.original.response)).length > 0; - const missingData = !hasMessages && !hasResponse; + const missingData = !hasMessages && !hasResponse && !hasError; // Format the response with error details if present const formattedResponse = () => { From c8d95470957492ff1dd0f6b5a6cf19bb7694d1df Mon Sep 17 00:00:00 2001 From: Varun Chawla <34209028+veeceey@users.noreply.github.com> Date: Sat, 7 Feb 2026 23:00:33 -0800 Subject: [PATCH 18/50] fix(ui): add null guard for models in API keys table (#20655) The VirtualKeysTable crashed when rendering keys with null or undefined models field. The className expression tried to access .length on null, throwing a TypeError that broke the entire keys table. Added Array.isArray() guard before accessing .length on the models value. Fixes #20611 --- .../VirtualKeysPage/VirtualKeysTable.test.tsx | 83 +++++++++++++++++++ .../VirtualKeysPage/VirtualKeysTable.tsx | 2 +- 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.test.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.test.tsx index ef5bb2a0371..749396c82f3 100644 --- a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.test.tsx +++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.test.tsx @@ -542,3 +542,86 @@ it("should display 'Default Proxy Admin' for created_by when value is 'default_u expect(defaultProxyAdminElements.length).toBeGreaterThan(0); }); }); + + +it("should render table without crashing when models is null", async () => { + const keyWithNullModels = { + ...mockKey, + models: null as unknown as string[], + }; + + mockUseFilterLogic.mockReturnValue({ + filters: { + "Team ID": "", + "Organization ID": "", + "Key Alias": "", + "User ID": "", + "Sort By": "created_at", + "Sort Order": "desc", + }, + filteredKeys: [keyWithNullModels], + allKeyAliases: ["test-key-alias"], + allTeams: [mockTeam], + allOrganizations: [mockOrganization], + handleFilterChange: vi.fn(), + handleFilterReset: vi.fn(), + }); + + const mockProps = { + teams: [mockTeam], + organizations: [mockOrganization], + onSortChange: vi.fn(), + currentSort: { + sortBy: "created_at", + sortOrder: "desc" as const, + }, + }; + + // This should not throw an error + renderWithProviders(); + + await waitFor(() => { + expect(screen.getByText("Test Key Alias")).toBeInTheDocument(); + }); +}); + +it("should render table without crashing when models is undefined", async () => { + const keyWithUndefinedModels = { + ...mockKey, + models: undefined as unknown as string[], + }; + + mockUseFilterLogic.mockReturnValue({ + filters: { + "Team ID": "", + "Organization ID": "", + "Key Alias": "", + "User ID": "", + "Sort By": "created_at", + "Sort Order": "desc", + }, + filteredKeys: [keyWithUndefinedModels], + allKeyAliases: ["test-key-alias"], + allTeams: [mockTeam], + allOrganizations: [mockOrganization], + handleFilterChange: vi.fn(), + handleFilterReset: vi.fn(), + }); + + const mockProps = { + teams: [mockTeam], + organizations: [mockOrganization], + onSortChange: vi.fn(), + currentSort: { + sortBy: "created_at", + sortOrder: "desc" as const, + }, + }; + + // This should not throw an error + renderWithProviders(); + + await waitFor(() => { + expect(screen.getByText("Test Key Alias")).toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx index 465b9b8fbe0..f7c47943e7e 100644 --- a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx +++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx @@ -727,7 +727,7 @@ export function VirtualKeysTable({ teams, organizations, onSortChange, currentSo whiteSpace: "pre-wrap", overflow: "hidden", }} - className={`py-0.5 max-h-8 overflow-hidden text-ellipsis whitespace-nowrap ${cell.column.id === "models" && (cell.getValue() as string[]).length > 3 ? "px-0" : ""}`} + className={`py-0.5 max-h-8 overflow-hidden text-ellipsis whitespace-nowrap ${cell.column.id === "models" && Array.isArray(cell.getValue()) && (cell.getValue() as string[]).length > 3 ? "px-0" : ""}`} > {flexRender(cell.column.columnDef.cell, cell.getContext())} From c9c6a5edc971c55cb5cc26cc5911faba6ee816aa Mon Sep 17 00:00:00 2001 From: Varun Chawla <34209028+veeceey@users.noreply.github.com> Date: Sat, 7 Feb 2026 23:02:29 -0800 Subject: [PATCH 19/50] Fix: Spend logs pickle error with Pydantic models and redaction (#20685) * docs: add callback registration optimization to v1.81.9 release notes (#20681) * docs: add callback registration optimization to v1.81.9 release notes * Update v1.81.9.md --------- Co-authored-by: Alexsander Hamir * Fix spend logs pickle error with Pydantic models Replace copy.deepcopy() with Pydantic-safe serialization to avoid "cannot pickle '_thread.RLock' object" errors when request/response redaction is enabled. Changes: - Add _convert_to_json_serializable_dict() helper that uses model_dump() for Pydantic models instead of pickle - Replace copy.deepcopy() calls in request and response redaction paths with the new helper function - Recursively handles nested dicts, lists, and Pydantic models Root cause: Pydantic v2 BaseModel instances contain internal _thread.RLock objects for thread-safety. When copy.deepcopy() attempts to pickle these objects, it fails because threading primitives cannot be pickled. Fixes #20647 * chore: remove unused copy import Remove unused copy import that was causing lint failure. The copy.deepcopy() calls were replaced with _convert_to_json_serializable_dict() helper function in the previous commit, making the copy module no longer needed. --------- Co-authored-by: ryan-crabbe <128659760+ryan-crabbe@users.noreply.github.com> Co-authored-by: Alexsander Hamir --- docs/my-website/release_notes/v1.81.9.md | 7 ++++ .../spend_tracking/spend_tracking_utils.py | 39 ++++++++++++++++--- 2 files changed, 40 insertions(+), 6 deletions(-) diff --git a/docs/my-website/release_notes/v1.81.9.md b/docs/my-website/release_notes/v1.81.9.md index 08b70e029e2..c34d3056cae 100644 --- a/docs/my-website/release_notes/v1.81.9.md +++ b/docs/my-website/release_notes/v1.81.9.md @@ -48,6 +48,13 @@ pip install litellm==1.81.9 - **UI Team Soft Budget Alerts** - [Set soft budgets on teams and receive email alerts when spending crosses the threshold β€” without blocking requests](../../docs/proxy/ui_team_soft_budget_alerts) - **Performance Optimizations** - Multiple performance improvements including ~40% Prometheus CPU reduction, LRU caching, and optimized logging paths - **LiteLLM Observatory** - [Automated 24-hour load tests](../../blog/litellm-observatory) +- **30% Faster Request Processing for Callback-Heavy Deployments** - [Performance improvement for callback heavy deployments][PR #20354](https://github.com/BerriAI/litellm/pull/20354) + +--- + +## 30% Faster Request Processing for Callback-Heavy Deployments + + If you use logging callbacks like Langfuse, Datadog, or Prometheus, every request was paying an unnecessary cost: three loops that re-sorted your callbacks on every single request, even though the callback list hadn't changed. The more callbacks you had configured, the more time was wasted. We moved this work to happen once at startup instead of on every request. For deployments with the default callback set, this is a ~30% speedup in request setup. For deployments with many callbacks configured, the improvement is even larger. --- diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py index bd148ecb481..cb8b9ec0395 100644 --- a/litellm/proxy/spend_tracking/spend_tracking_utils.py +++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py @@ -1,4 +1,3 @@ -import copy import hashlib import json import secrets @@ -642,6 +641,34 @@ def _sanitize_request_body_for_spend_logs_payload( return {k: _sanitize_value(v) for k, v in request_body.items()} +def _convert_to_json_serializable_dict(obj: Any) -> Any: + """ + Convert object to JSON-serializable dict, handling Pydantic models safely. + + This avoids pickle-based deepcopy which fails on Pydantic v2 models + containing _thread.RLock objects. + + Args: + obj: Object to convert (dict, list, Pydantic model, or primitive) + + Returns: + JSON-serializable version of the object + """ + if isinstance(obj, BaseModel): + # Use Pydantic's model_dump() instead of pickle + return obj.model_dump() + elif isinstance(obj, dict): + return {k: _convert_to_json_serializable_dict(v) for k, v in obj.items()} + elif isinstance(obj, list): + return [_convert_to_json_serializable_dict(item) for item in obj] + elif hasattr(obj, "__dict__"): + # Handle objects with __dict__ attribute + return _convert_to_json_serializable_dict(obj.__dict__) + else: + # Primitives (str, int, float, bool, None) pass through + return obj + + def _get_proxy_server_request_for_spend_logs_payload( metadata: dict, litellm_params: dict, @@ -649,7 +676,7 @@ def _get_proxy_server_request_for_spend_logs_payload( ) -> str: """ Only store if _should_store_prompts_and_responses_in_spend_logs() is True - + If turn_off_message_logging is enabled, redact messages in the request body. """ if _should_store_prompts_and_responses_in_spend_logs(): @@ -674,9 +701,9 @@ def _get_proxy_server_request_for_spend_logs_payload( ), } - # If redaction is enabled, deep copy request body before redacting + # If redaction is enabled, convert to serializable dict before redacting if should_redact_message_logging(model_call_details=model_call_details): - _request_body = copy.deepcopy(_request_body) + _request_body = _convert_to_json_serializable_dict(_request_body) perform_redaction(model_call_details=_request_body, result=None) _request_body = _sanitize_request_body_for_spend_logs_payload(_request_body) @@ -736,9 +763,9 @@ def _get_response_for_spend_logs_payload( ), } - # If redaction is enabled, deep copy response before redacting + # If redaction is enabled, convert to serializable dict before redacting if should_redact_message_logging(model_call_details=model_call_details): - response_obj = copy.deepcopy(response_obj) + response_obj = _convert_to_json_serializable_dict(response_obj) response_obj = perform_redaction(model_call_details={}, result=response_obj) sanitized_wrapper = _sanitize_request_body_for_spend_logs_payload( From 0458e734b2642aaf25fb381da423bdf0e71363e2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Elias=20H=C3=B6gbom=20Aronsson?= Date: Sun, 8 Feb 2026 08:05:17 +0100 Subject: [PATCH 20/50] fix(vertex_ai): propagate extra_headers anthropic-beta to request body (#20666) Vertex AI requires Anthropic beta flags in the request body (anthropic_beta array), not as HTTP headers. The Bedrock handler already extracts user-specified beta headers from the headers dict, but the Vertex handler was missing this, causing extra_headers like interleaved-thinking-2025-05-14 to be silently dropped. This extracts anthropic-beta values from optional_params extra_headers and merges them into the anthropic_beta request body field, and also removes extra_headers from the request body since the parent's transform_request spreads optional_params into data. --- .../anthropic/transformation.py | 46 ++- ...partner_models_anthropic_transformation.py | 341 ++++++++++++------ 2 files changed, 256 insertions(+), 131 deletions(-) diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 0b728d88e76..6a5b934661a 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -56,34 +56,36 @@ class VertexAIAnthropicConfig(AnthropicConfig): ) -> None: """ Add context_management beta headers to the beta_set. - + - If any edit has type "compact_20260112", add compact-2026-01-12 header - For all other edits, add context-management-2025-06-27 header - + Args: beta_set: Set of beta headers to modify in-place context_management: The context_management dict from optional_params """ from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES - + edits = context_management.get("edits", []) has_compact = False has_other = False - + for edit in edits: edit_type = edit.get("type", "") if edit_type == "compact_20260112": has_compact = True else: has_other = True - + # Add compact header if any compact edits exist if has_compact: beta_set.add(ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_01_12.value) - + # Add context management header if any other edits exist if has_other: - beta_set.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value) + beta_set.add( + ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value + ) def transform_request( self, @@ -102,10 +104,10 @@ class VertexAIAnthropicConfig(AnthropicConfig): ) data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter - + # VertexAI doesn't support output_format parameter, remove it if present data.pop("output_format", None) - + tools = optional_params.get("tools") tool_search_used = self.is_tool_search_used(tools) auto_betas = self.get_anthropic_beta_list( @@ -119,16 +121,30 @@ class VertexAIAnthropicConfig(AnthropicConfig): beta_set = set(auto_betas) if tool_search_used: - beta_set.add("tool-search-tool-2025-10-19") # Vertex requires this header for tool search - + beta_set.add( + "tool-search-tool-2025-10-19" + ) # Vertex requires this header for tool search + # Add context_management beta headers (compact and/or context-management) context_management = optional_params.get("context_management") if context_management: self._add_context_management_beta_headers(beta_set, context_management) + extra_headers = optional_params.get("extra_headers") or {} + anthropic_beta_value = extra_headers.get("anthropic-beta", "") + if isinstance(anthropic_beta_value, str) and anthropic_beta_value: + for beta in anthropic_beta_value.split(","): + beta = beta.strip() + if beta: + beta_set.add(beta) + elif isinstance(anthropic_beta_value, list): + beta_set.update(anthropic_beta_value) + + data.pop("extra_headers", None) + if beta_set: data["anthropic_beta"] = list(beta_set) - + return data def map_openai_params( @@ -148,7 +164,7 @@ class VertexAIAnthropicConfig(AnthropicConfig): original_model = model if "response_format" in non_default_params: model = "claude-3-sonnet-20240229" # Use a model that will use tool-based approach - + # Call parent method with potentially modified model name optional_params = super().map_openai_params( non_default_params=non_default_params, @@ -156,10 +172,10 @@ class VertexAIAnthropicConfig(AnthropicConfig): model=model, drop_params=drop_params, ) - + # Restore original model name for any other processing model = original_model - + return optional_params def transform_response( diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 90ab41aadf6..4bcafd4c57e 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -45,68 +45,65 @@ def test_vertex_ai_anthropic_web_search_header_in_completion(): # Create the config instance model_info = AnthropicModelInfo() - + # Test the header generation directly tools = [{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}] - + # Check if web search tool is detected web_search_detected = model_info.is_web_search_tool_used(tools=tools) assert web_search_detected is True, "Web search tool should be detected" - + # Generate headers with is_vertex_request=True headers = model_info.get_anthropic_headers( api_key="test-key", web_search_tool_used=web_search_detected, is_vertex_request=True, ) - + # Assert that the anthropic-beta header with web-search is present assert "anthropic-beta" in headers, "anthropic-beta header should be present" - assert headers["anthropic-beta"] == "web-search-2025-03-05", \ - f"anthropic-beta should be 'web-search-2025-03-05', got: {headers['anthropic-beta']}" - + assert ( + headers["anthropic-beta"] == "web-search-2025-03-05" + ), f"anthropic-beta should be 'web-search-2025-03-05', got: {headers['anthropic-beta']}" + # Test that header is NOT added for non-Vertex requests headers_non_vertex = model_info.get_anthropic_headers( api_key="test-key", web_search_tool_used=web_search_detected, is_vertex_request=False, ) - + # For non-Vertex (Anthropic-hosted), the web search header should NOT be in anthropic-beta # because Anthropic doesn't require it - assert "anthropic-beta" not in headers_non_vertex or "web-search" not in headers_non_vertex.get("anthropic-beta", ""), \ - "anthropic-beta with web-search should not be present for non-Vertex requests" + assert ( + "anthropic-beta" not in headers_non_vertex + or "web-search" not in headers_non_vertex.get("anthropic-beta", "") + ), "anthropic-beta with web-search should not be present for non-Vertex requests" def test_vertex_ai_anthropic_context_management_compact_beta_header(): """Test that context_management with compact adds the correct beta header for Vertex AI""" config = VertexAIAnthropicConfig() - + messages = [{"role": "user", "content": "Hello"}] optional_params = { - "context_management": { - "edits": [ - { - "type": "compact_20260112" - } - ] - }, + "context_management": {"edits": [{"type": "compact_20260112"}]}, "max_tokens": 100, - "is_vertex_request": True + "is_vertex_request": True, } - + result = config.transform_request( model="claude-opus-4-6", messages=messages, optional_params=optional_params, litellm_params={}, - headers={} + headers={}, ) - + # Verify context_management is included assert "context_management" in result assert result["context_management"]["edits"][0]["type"] == "compact_20260112" - + # Verify compact beta header is in anthropic_beta field assert "anthropic_beta" in result assert "compact-2026-01-12" in result["anthropic_beta"] @@ -115,33 +112,27 @@ def test_vertex_ai_anthropic_context_management_compact_beta_header(): def test_vertex_ai_anthropic_context_management_mixed_edits(): """Test that context_management with both compact and other edits adds both beta headers""" config = VertexAIAnthropicConfig() - + messages = [{"role": "user", "content": "Hello"}] optional_params = { "context_management": { "edits": [ - { - "type": "compact_20260112" - }, - { - "type": "replace", - "message_id": "msg_123", - "content": "new content" - } + {"type": "compact_20260112"}, + {"type": "replace", "message_id": "msg_123", "content": "new content"}, ] }, "max_tokens": 100, - "is_vertex_request": True + "is_vertex_request": True, } - + result = config.transform_request( model="claude-opus-4-6", messages=messages, optional_params=optional_params, litellm_params={}, - headers={} + headers={}, ) - + # Verify both beta headers are present assert "anthropic_beta" in result assert "compact-2026-01-12" in result["anthropic_beta"] @@ -151,58 +142,65 @@ def test_vertex_ai_anthropic_context_management_mixed_edits(): def test_vertex_ai_anthropic_structured_output_header_not_added(): """Test that structured output beta headers are NOT added for Vertex AI requests""" from litellm.llms.anthropic.chat.transformation import AnthropicConfig - + config = AnthropicConfig() - + # Test case 1: Vertex request with output_format should NOT add beta header headers_vertex = {} optional_params_vertex = { - 'output_format': { - 'type': 'json_schema', - 'json_schema': { - 'name': 'MathResult', - 'schema': {'properties': {'result': {'type': 'integer'}}} - } + "output_format": { + "type": "json_schema", + "json_schema": { + "name": "MathResult", + "schema": {"properties": {"result": {"type": "integer"}}}, + }, }, - 'is_vertex_request': True + "is_vertex_request": True, } - result_vertex = config.update_headers_with_optional_anthropic_beta(headers_vertex, optional_params_vertex) - - assert "anthropic-beta" not in result_vertex, \ - f"Vertex request should NOT have anthropic-beta header for structured output, got: {result_vertex.get('anthropic-beta')}" - + result_vertex = config.update_headers_with_optional_anthropic_beta( + headers_vertex, optional_params_vertex + ) + + assert ( + "anthropic-beta" not in result_vertex + ), f"Vertex request should NOT have anthropic-beta header for structured output, got: {result_vertex.get('anthropic-beta')}" + # Test case 2: Non-Vertex request with output_format SHOULD add beta header headers_non_vertex = {} optional_params_non_vertex = { - 'output_format': { - 'type': 'json_schema', - 'json_schema': { - 'name': 'MathResult', - 'schema': {'properties': {'result': {'type': 'integer'}}} - } + "output_format": { + "type": "json_schema", + "json_schema": { + "name": "MathResult", + "schema": {"properties": {"result": {"type": "integer"}}}, + }, }, - 'is_vertex_request': False + "is_vertex_request": False, } - result_non_vertex = config.update_headers_with_optional_anthropic_beta(headers_non_vertex, optional_params_non_vertex) - - assert "anthropic-beta" in result_non_vertex, \ - "Non-Vertex request SHOULD have anthropic-beta header for structured output" - assert result_non_vertex["anthropic-beta"] == "structured-outputs-2025-11-13", \ - f"Expected 'structured-outputs-2025-11-13', got: {result_non_vertex.get('anthropic-beta')}" + result_non_vertex = config.update_headers_with_optional_anthropic_beta( + headers_non_vertex, optional_params_non_vertex + ) + + assert ( + "anthropic-beta" in result_non_vertex + ), "Non-Vertex request SHOULD have anthropic-beta header for structured output" + assert ( + result_non_vertex["anthropic-beta"] == "structured-outputs-2025-11-13" + ), f"Expected 'structured-outputs-2025-11-13', got: {result_non_vertex.get('anthropic-beta')}" def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): """ - Test fix for issue #18625: Claude Sonnet 4.5 on VertexAI should use tool-based + Test fix for issue #18625: Claude Sonnet 4.5 on VertexAI should use tool-based structured outputs instead of output_format parameter. - + This test verifies that: 1. Claude Sonnet 4.5 uses tool-based structured outputs on VertexAI 2. output_format parameter is removed from the final request 3. The fix prevents "Extra inputs are not permitted" error """ config = VertexAIAnthropicConfig() - + # Test data matching the issue report response_format = { "type": "json_schema", @@ -212,29 +210,23 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): "schema": { "type": "object", "properties": { - "question": { - "type": "string" - }, - "response": { - "type": "string" - } + "question": {"type": "string"}, + "response": {"type": "string"}, }, "required": ["question", "response"], - "additionalProperties": False - } - } + "additionalProperties": False, + }, + }, } - - messages = [ - {"role": "user", "content": "Generate a question and answer about AI."} - ] - + + messages = [{"role": "user", "content": "Generate a question and answer about AI."}] + # Test parameters that would trigger the issue non_default_params = { "response_format": response_format, "max_tokens": 1000, } - + # Test 1: Verify map_openai_params forces tool-based approach for Claude Sonnet 4.5 optional_params = {} result_params = config.map_openai_params( @@ -243,17 +235,19 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): model="claude-3-5-sonnet-20241022", # Claude Sonnet 4.5 model drop_params=False, ) - + # Should have tools and tool_choice (tool-based approach) assert "tools" in result_params, "Tools should be present for structured output" - assert "tool_choice" in result_params, "Tool choice should be present for structured output" + assert ( + "tool_choice" in result_params + ), "Tool choice should be present for structured output" assert "json_mode" in result_params, "JSON mode should be enabled" - + # Verify the tool is the response format tool tools = result_params["tools"] assert len(tools) == 1, "Should have exactly one tool for response format" assert tools[0]["name"] == "json_tool_call", "Tool should be named json_tool_call" - + # Test 2: Verify transform_request removes output_format parameter # Simulate what would happen if parent class added output_format test_data = { @@ -264,20 +258,22 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): "tool_choice": result_params["tool_choice"], "output_format": { # This would be added by parent class for Sonnet 4.5 "type": "json_schema", - "schema": response_format["json_schema"]["schema"] - } + "schema": response_format["json_schema"]["schema"], + }, } - + # Mock the parent transform_request to return data with output_format original_transform = config.__class__.__bases__[0].transform_request - - def mock_transform_request(self, model, messages, optional_params, litellm_params, headers): + + def mock_transform_request( + self, model, messages, optional_params, litellm_params, headers + ): # Return test data that includes output_format return test_data.copy() - + # Temporarily replace parent method config.__class__.__bases__[0].transform_request = mock_transform_request - + try: final_data = config.transform_request( model="claude-3-5-sonnet-20241022", @@ -286,13 +282,15 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): litellm_params={}, headers={}, ) - + # Verify that output_format was removed (fixes the "Extra inputs are not permitted" error) - assert "output_format" not in final_data, "output_format should be removed for VertexAI" + assert ( + "output_format" not in final_data + ), "output_format should be removed for VertexAI" assert "model" not in final_data, "model should be removed for VertexAI" assert "tools" in final_data, "tools should still be present" assert "tool_choice" in final_data, "tool_choice should still be present" - + finally: # Restore original method config.__class__.__bases__[0].transform_request = original_transform @@ -300,43 +298,149 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): def test_vertex_ai_anthropic_other_models_still_use_tools(): """ - Test that other Anthropic models (non-Sonnet 4.5) on VertexAI also use tool-based + Test that other Anthropic models (non-Sonnet 4.5) on VertexAI also use tool-based structured outputs, ensuring consistency across all models. """ config = VertexAIAnthropicConfig() - + response_format = { "type": "json_schema", "json_schema": { "name": "test_schema", - "schema": { - "type": "object", - "properties": { - "result": {"type": "string"} - } - } - } + "schema": {"type": "object", "properties": {"result": {"type": "string"}}}, + }, } - + # Test with Claude 3 Sonnet (not 4.5) non_default_params = {"response_format": response_format} optional_params = {} - + result_params = config.map_openai_params( non_default_params=non_default_params, optional_params=optional_params, model="claude-3-sonnet-20240229", drop_params=False, ) - + # Should still use tool-based approach - assert "tools" in result_params, "Claude 3 Sonnet should also use tool-based structured output" + assert ( + "tools" in result_params + ), "Claude 3 Sonnet should also use tool-based structured output" assert "tool_choice" in result_params, "Tool choice should be present" assert "json_mode" in result_params, "JSON mode should be enabled" + +def test_vertex_ai_anthropic_extra_headers_beta_propagation(): + """Test that anthropic-beta values from extra_headers are propagated to the + anthropic_beta request body field for Vertex AI requests. + + Vertex AI requires beta flags in the request body (anthropic_beta array), + not as HTTP headers. This mirrors the Bedrock handler's behavior of + extracting user-specified beta headers. + """ + config = VertexAIAnthropicConfig() + + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "max_tokens": 100, + "is_vertex_request": True, + "extra_headers": { + "anthropic-beta": "interleaved-thinking-2025-05-14", + }, + } + + result = config.transform_request( + model="claude-sonnet-4-20250514", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "anthropic_beta" in result + assert "interleaved-thinking-2025-05-14" in result["anthropic_beta"] + assert "extra_headers" not in result + + +def test_vertex_ai_anthropic_extra_headers_beta_merged_with_auto_betas(): + """Test that extra_headers betas are merged with auto-detected betas + rather than replacing them.""" + config = VertexAIAnthropicConfig() + + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "max_tokens": 100, + "is_vertex_request": True, + "extra_headers": { + "anthropic-beta": "interleaved-thinking-2025-05-14", + }, + "context_management": {"edits": [{"type": "compact_20260112"}]}, + } + + result = config.transform_request( + model="claude-opus-4-6", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "anthropic_beta" in result + assert "interleaved-thinking-2025-05-14" in result["anthropic_beta"] + assert "compact-2026-01-12" in result["anthropic_beta"] + + +def test_vertex_ai_anthropic_extra_headers_comma_separated_betas(): + """Test that comma-separated beta values in extra_headers are all extracted.""" + config = VertexAIAnthropicConfig() + + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "max_tokens": 100, + "is_vertex_request": True, + "extra_headers": { + "anthropic-beta": "interleaved-thinking-2025-05-14,dev-full-thinking-2025-05-14", + }, + } + + result = config.transform_request( + model="claude-sonnet-4-20250514", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "anthropic_beta" in result + assert "interleaved-thinking-2025-05-14" in result["anthropic_beta"] + assert "dev-full-thinking-2025-05-14" in result["anthropic_beta"] + + +def test_vertex_ai_anthropic_no_extra_headers_unchanged(): + """Test that requests without extra_headers still work normally.""" + config = VertexAIAnthropicConfig() + + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "max_tokens": 100, + "is_vertex_request": True, + } + + result = config.transform_request( + model="claude-sonnet-4-20250514", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "anthropic_beta" not in result + assert "extra_headers" not in result + + def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_header(): """ - Test that remove_unsupported_beta correctly filters out prompt-caching-scope-2026-01-05 + Test that remove_unsupported_beta correctly filters out prompt-caching-scope-2026-01-05 from the anthropic-beta headers. """ from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.experimental_pass_through.transformation import ( @@ -352,13 +456,18 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea headers = update_headers_with_filtered_beta(headers, "vertex_ai") beta_header = headers.get("anthropic-beta") - assert PROMPT_CACHING_BETA_HEADER not in (beta_header or ""), \ - f"{PROMPT_CACHING_BETA_HEADER} should be filtered out" - assert "other-feature" in (beta_header or ""), \ - "Other non-excluded beta headers should remain" - assert "web-search-2025-03-05" in (beta_header or ""), \ - "Other non-excluded beta headers should remain" + assert PROMPT_CACHING_BETA_HEADER not in ( + beta_header or "" + ), f"{PROMPT_CACHING_BETA_HEADER} should be filtered out" + assert "other-feature" in ( + beta_header or "" + ), "Other non-excluded beta headers should remain" + assert "web-search-2025-03-05" in ( + beta_header or "" + ), "Other non-excluded beta headers should remain" # If prompt-caching was the only value, header should be removed completely headers2 = {"anthropic-beta": PROMPT_CACHING_BETA_HEADER} headers2 = update_headers_with_filtered_beta(headers2, "vertex_ai") - assert "anthropic-beta" not in headers2, "Header should be removed if no supported values remain" \ No newline at end of file + assert ( + "anthropic-beta" not in headers2 + ), "Header should be removed if no supported values remain" From f7d03f8a43fec1cc53358483e9bb30d2320403e5 Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Sun, 8 Feb 2026 05:04:47 -0600 Subject: [PATCH 21/50] fix(streaming): preserve interleaved thinking/redacted blocks --- .../streaming_chunk_builder_utils.py | 58 ++++++++------- .../test_streaming_chunk_builder_utils.py | 74 ++++++++++++++++++- 2 files changed, 104 insertions(+), 28 deletions(-) diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index 53252df0a28..76c7246b87e 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -1,6 +1,6 @@ import base64 import time -from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Union, cast +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union, cast from litellm.types.llms.openai import ( ChatCompletionAssistantContentValue, @@ -326,10 +326,22 @@ class ChunkProcessor: thinking_blocks: List[ Union["ChatCompletionThinkingBlock", "ChatCompletionRedactedThinkingBlock"] ] = [] - combined_thinking_text: Optional[str] = None - data: Optional[str] = None - signature: Optional[str] = None - type: Literal["thinking", "redacted_thinking"] = "thinking" + current_thinking_text_parts: List[str] = [] + current_signature: Optional[str] = None + + def _flush_thinking_block() -> None: + nonlocal current_thinking_text_parts, current_signature + if len(current_thinking_text_parts) > 0 and current_signature: + thinking_blocks.append( + ChatCompletionThinkingBlock( + type="thinking", + thinking="".join(current_thinking_text_parts), + signature=current_signature, + ) + ) + current_thinking_text_parts = [] + current_signature = None + for chunk in chunks: choices = chunk["choices"] for choice in choices: @@ -339,33 +351,25 @@ class ChunkProcessor: for thinking_block in thinking: thinking_type = thinking_block.get("type", None) if thinking_type and thinking_type == "redacted_thinking": - type = "redacted_thinking" - data = thinking_block.get("data", None) + _flush_thinking_block() + redacted_data = thinking_block.get("data", None) + if redacted_data: + thinking_blocks.append( + ChatCompletionRedactedThinkingBlock( + type="redacted_thinking", + data=redacted_data, + ) + ) else: - type = "thinking" thinking_text = thinking_block.get("thinking", None) if thinking_text: - if combined_thinking_text is None: - combined_thinking_text = "" - - combined_thinking_text += thinking_text + current_thinking_text_parts.append(thinking_text) signature = thinking_block.get("signature", None) + if signature: + current_signature = signature + _flush_thinking_block() - if combined_thinking_text and type == "thinking" and signature: - thinking_blocks.append( - ChatCompletionThinkingBlock( - type=type, - thinking=combined_thinking_text, - signature=signature, - ) - ) - elif data and type == "redacted_thinking": - thinking_blocks.append( - ChatCompletionRedactedThinkingBlock( - type=type, - data=data, - ) - ) + _flush_thinking_block() if len(thinking_blocks) > 0: return thinking_blocks diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py index eef206ca667..34c9efd8382 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py @@ -158,6 +158,78 @@ def test_get_combined_tool_content(): ] +def test_get_combined_thinking_content_preserves_interleaved_blocks(): + base_chunk = { + "id": "chatcmpl-123", + "object": "chat.completion.chunk", + "created": 1234567890, + "model": "claude-sonnet-4-20250514", + } + + def make_chunk(**delta_kwargs): + return ModelResponseStream( + **{ + **base_chunk, + "choices": [ + { + "index": 0, + "delta": delta_kwargs, + "finish_reason": None, + } + ], + } + ) + + chunks = [ + make_chunk(role="assistant", content=None), + make_chunk( + thinking_blocks=[ + {"type": "thinking", "thinking": "Step 1 analysis...", "signature": None} + ] + ), + make_chunk( + thinking_blocks=[ + {"type": "thinking", "thinking": None, "signature": "sig_block1"} + ] + ), + make_chunk( + thinking_blocks=[ + { + "type": "redacted_thinking", + "data": "EuoBCoYBGAIi...encrypted...", + } + ] + ), + make_chunk( + thinking_blocks=[ + {"type": "thinking", "thinking": "Step 2 analysis...", "signature": None} + ] + ), + make_chunk( + thinking_blocks=[ + {"type": "thinking", "thinking": None, "signature": "sig_block2"} + ] + ), + ] + + thinking_chunks = [ + chunk for chunk in chunks if chunk["choices"][0]["delta"].get("thinking_blocks") + ] + processor = ChunkProcessor(chunks=chunks) + result = processor.get_combined_thinking_content(thinking_chunks) + + assert result is not None + assert len(result) == 3 + assert result[0]["type"] == "thinking" + assert result[0]["thinking"] == "Step 1 analysis..." + assert result[0]["signature"] == "sig_block1" + assert result[1]["type"] == "redacted_thinking" + assert result[1]["data"] == "EuoBCoYBGAIi...encrypted..." + assert result[2]["type"] == "thinking" + assert result[2]["thinking"] == "Step 2 analysis..." + assert result[2]["signature"] == "sig_block2" + + def test_cache_read_input_tokens_retained(): chunk1 = ModelResponseStream( id="chatcmpl-95aabb85-c39f-443d-ae96-0370c404d70c", @@ -441,4 +513,4 @@ def test_stream_chunk_builder_anthropic_web_search(): assert usage.prompt_tokens == 50 assert usage.completion_tokens == 27 assert usage.total_tokens == 77 - assert usage.server_tool_use['web_search_requests'] == 2 \ No newline at end of file + assert usage.server_tool_use['web_search_requests'] == 2 From c63d5fa0b5bc165dd6ae11a50b22358262078dbd Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Sun, 8 Feb 2026 05:08:50 -0600 Subject: [PATCH 22/50] test(streaming): build thinking chunks with typed Delta/StreamingChoices --- .../test_streaming_chunk_builder_utils.py | 18 ++++++++---------- 1 file changed, 8 insertions(+), 10 deletions(-) diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py index 34c9efd8382..da6d8027921 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py @@ -168,16 +168,14 @@ def test_get_combined_thinking_content_preserves_interleaved_blocks(): def make_chunk(**delta_kwargs): return ModelResponseStream( - **{ - **base_chunk, - "choices": [ - { - "index": 0, - "delta": delta_kwargs, - "finish_reason": None, - } - ], - } + **base_chunk, + choices=[ + StreamingChoices( + index=0, + delta=Delta(**delta_kwargs), + finish_reason=None, + ) + ], ) chunks = [ From 381c3756f45eafa8fbdbbf774b694cc99da02773 Mon Sep 17 00:00:00 2001 From: tshushan Date: Sun, 8 Feb 2026 16:24:14 +0200 Subject: [PATCH 23/50] Fix video list pagination cursors not encoded with provider metadata first_id and last_id in the video list response were returned as raw provider IDs while data[].id was properly wrapped with encode_video_id_with_provider(). This caused pagination to break when clients passed unencoded cursors back as the `after` parameter. - Encode first_id/last_id in transform_video_list_response - Decode the `after` param in transform_video_list_request via extract_original_video_id() - Add 6 unit tests covering encoding, decoding, passthrough, and full round-trip pagination Fixes #20708 Co-Authored-By: Claude Opus 4.6 --- litellm/llms/openai/videos/transformation.py | 45 +++-- tests/test_litellm/test_video_generation.py | 175 +++++++++++++++++++ 2 files changed, 209 insertions(+), 11 deletions(-) diff --git a/litellm/llms/openai/videos/transformation.py b/litellm/llms/openai/videos/transformation.py index 3073b22e1ca..0dd7940a92e 100644 --- a/litellm/llms/openai/videos/transformation.py +++ b/litellm/llms/openai/videos/transformation.py @@ -269,26 +269,27 @@ class OpenAIVideoConfig(BaseVideoConfig): ) -> Tuple[str, Dict]: """ Transform the video list request for OpenAI API. - + OpenAI API expects the following request: - GET /v1/videos """ # Use the api_base directly for video list url = api_base - + # Prepare query parameters params = {} if after is not None: - params["after"] = after + # Decode the wrapped video ID back to the original provider ID + params["after"] = extract_original_video_id(after) if limit is not None: params["limit"] = str(limit) if order is not None: params["order"] = order - + # Add any extra query parameters if extra_query: params.update(extra_query) - + return url, params def transform_video_list_response( @@ -296,18 +297,40 @@ class OpenAIVideoConfig(BaseVideoConfig): raw_response: httpx.Response, logging_obj: LiteLLMLoggingObj, custom_llm_provider: Optional[str] = None, - ) -> Dict[str,str]: + ) -> Dict[str, str]: response_data = raw_response.json() - + if custom_llm_provider and "data" in response_data: for video_obj in response_data.get("data", []): if isinstance(video_obj, dict) and "id" in video_obj: video_obj["id"] = encode_video_id_with_provider( - video_obj["id"], - custom_llm_provider, - video_obj.get("model") + video_obj["id"], + custom_llm_provider, + video_obj.get("model"), ) - + + # Encode pagination cursor IDs so they remain consistent + # with the wrapped data[].id format + data_list = response_data.get("data", []) + if response_data.get("first_id"): + first_model = None + if data_list and isinstance(data_list[0], dict): + first_model = data_list[0].get("model") + response_data["first_id"] = encode_video_id_with_provider( + response_data["first_id"], + custom_llm_provider, + first_model, + ) + if response_data.get("last_id"): + last_model = None + if data_list and isinstance(data_list[-1], dict): + last_model = data_list[-1].get("model") + response_data["last_id"] = encode_video_id_with_provider( + response_data["last_id"], + custom_llm_provider, + last_model, + ) + return response_data def transform_video_delete_request( diff --git a/tests/test_litellm/test_video_generation.py b/tests/test_litellm/test_video_generation.py index cfc1535052c..5446a0a7b3f 100644 --- a/tests/test_litellm/test_video_generation.py +++ b/tests/test_litellm/test_video_generation.py @@ -916,6 +916,181 @@ def test_encode_video_id_with_provider_handles_azure_video_prefix(): ) assert encoded_twice == encoded_id # Should return the same encoded ID +class TestVideoListTransformation: + """Tests for video list request/response transformation with provider ID encoding.""" + + def test_transform_video_list_response_encodes_first_id_and_last_id(self): + """Verify that first_id and last_id are encoded with provider metadata.""" + config = OpenAIVideoConfig() + + mock_http_response = MagicMock() + mock_http_response.json.return_value = { + "object": "list", + "data": [ + { + "id": "video_aaa", + "object": "video", + "model": "sora-2", + "status": "completed", + }, + { + "id": "video_bbb", + "object": "video", + "model": "sora-2", + "status": "completed", + }, + ], + "first_id": "video_aaa", + "last_id": "video_bbb", + "has_more": False, + } + + result = config.transform_video_list_response( + raw_response=mock_http_response, + logging_obj=MagicMock(), + custom_llm_provider="azure", + ) + + from litellm.types.videos.utils import decode_video_id_with_provider + + # data[].id should be encoded + for item in result["data"]: + decoded = decode_video_id_with_provider(item["id"]) + assert decoded["custom_llm_provider"] == "azure" + + # first_id and last_id should also be encoded + first_decoded = decode_video_id_with_provider(result["first_id"]) + assert first_decoded["custom_llm_provider"] == "azure" + assert first_decoded["video_id"] == "video_aaa" + assert first_decoded["model_id"] == "sora-2" + + last_decoded = decode_video_id_with_provider(result["last_id"]) + assert last_decoded["custom_llm_provider"] == "azure" + assert last_decoded["video_id"] == "video_bbb" + assert last_decoded["model_id"] == "sora-2" + + def test_transform_video_list_response_no_provider_leaves_ids_unchanged(self): + """When custom_llm_provider is None, all IDs should remain unchanged.""" + config = OpenAIVideoConfig() + + mock_http_response = MagicMock() + mock_http_response.json.return_value = { + "object": "list", + "data": [ + {"id": "video_aaa", "object": "video", "model": "sora-2", "status": "completed"}, + ], + "first_id": "video_aaa", + "last_id": "video_aaa", + "has_more": False, + } + + result = config.transform_video_list_response( + raw_response=mock_http_response, + logging_obj=MagicMock(), + custom_llm_provider=None, + ) + + assert result["data"][0]["id"] == "video_aaa" + assert result["first_id"] == "video_aaa" + assert result["last_id"] == "video_aaa" + + def test_transform_video_list_response_missing_pagination_fields(self): + """first_id / last_id may be absent or null; should not raise.""" + config = OpenAIVideoConfig() + + mock_http_response = MagicMock() + mock_http_response.json.return_value = { + "object": "list", + "data": [ + {"id": "video_aaa", "object": "video", "model": "sora-2", "status": "completed"}, + ], + "has_more": False, + } + + result = config.transform_video_list_response( + raw_response=mock_http_response, + logging_obj=MagicMock(), + custom_llm_provider="azure", + ) + + # data[].id should still be encoded + from litellm.types.videos.utils import decode_video_id_with_provider + + decoded = decode_video_id_with_provider(result["data"][0]["id"]) + assert decoded["custom_llm_provider"] == "azure" + + # first_id / last_id should not be present + assert "first_id" not in result + assert "last_id" not in result + + def test_transform_video_list_request_decodes_after_parameter(self): + """Encoded 'after' cursor should be decoded back to the raw provider ID.""" + from litellm.types.videos.utils import encode_video_id_with_provider + + config = OpenAIVideoConfig() + + raw_id = "video_69888baee890819086dd3366bfc372fe" + encoded_id = encode_video_id_with_provider(raw_id, "azure", "sora-2") + + url, params = config.transform_video_list_request( + api_base="https://my-resource.openai.azure.com/openai/v1/videos", + litellm_params=MagicMock(), + headers={}, + after=encoded_id, + limit=10, + ) + + assert params["after"] == raw_id + assert params["limit"] == "10" + + def test_transform_video_list_request_passes_through_plain_after(self): + """A plain (non-encoded) 'after' value should pass through unchanged.""" + config = OpenAIVideoConfig() + + url, params = config.transform_video_list_request( + api_base="https://api.openai.com/v1/videos", + litellm_params=MagicMock(), + headers={}, + after="video_plain_id", + ) + + assert params["after"] == "video_plain_id" + + def test_transform_video_list_roundtrip(self): + """first_id from list response should decode correctly when used as after parameter.""" + config = OpenAIVideoConfig() + + # Simulate a list response + mock_http_response = MagicMock() + mock_http_response.json.return_value = { + "object": "list", + "data": [ + {"id": "video_aaa", "object": "video", "model": "sora-2", "status": "completed"}, + {"id": "video_bbb", "object": "video", "model": "sora-2", "status": "completed"}, + ], + "first_id": "video_aaa", + "last_id": "video_bbb", + "has_more": True, + } + + list_result = config.transform_video_list_response( + raw_response=mock_http_response, + logging_obj=MagicMock(), + custom_llm_provider="azure", + ) + + # Use the encoded last_id as the 'after' cursor for the next page + _, params = config.transform_video_list_request( + api_base="https://my-resource.openai.azure.com/openai/v1/videos", + litellm_params=MagicMock(), + headers={}, + after=list_result["last_id"], + ) + + # The after param sent to the upstream API should be the raw video ID + assert params["after"] == "video_bbb" + + class TestVideoEndpointsProxyLitellmParams: """Test that video proxy endpoints (status, content, remix) respect litellm_params from proxy config.""" From 68d788c84d761b6a94d73480e43e6990f5eb4a19 Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Sun, 8 Feb 2026 08:48:39 -0600 Subject: [PATCH 24/50] fix(responses): preserve streamed tool deltas when id is omitted --- .../streaming_iterator.py | 29 ++++- ...test_tool_call_streaming_transformation.py | 102 ++++++++++++++++++ 2 files changed, 129 insertions(+), 2 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index 867c18b6dd4..ea9b8889d39 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -88,6 +88,7 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self._pending_tool_events: List[BaseLiteLLMOpenAIResponseObject] = [] self._tool_output_index_by_call_id: dict[str, int] = {} self._tool_args_by_call_id: dict[str, str] = {} + self._tool_call_id_by_index: dict[int, str] = {} self._next_tool_output_index: int = 1 # output_index=0 reserved for the message item self._final_tool_events_queued: bool = False self._sequence_number: int = 0 @@ -111,6 +112,19 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self._tool_output_index_by_call_id[call_id] = idx return idx + def _normalize_tool_call_index(self, tool_call: object) -> Optional[int]: + idx_raw = ( + tool_call.get("index") + if isinstance(tool_call, dict) + else getattr(tool_call, "index", None) + ) + if idx_raw is None: + return None + try: + return int(idx_raw) + except (TypeError, ValueError): + return None + def _is_reasoning_end(self, chunk): delta = chunk.choices[0].delta @@ -143,10 +157,21 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): return for tc in tool_calls: + tc_index = self._normalize_tool_call_index(tc) call_id_raw = tc.get("id") if isinstance(tc, dict) else getattr(tc, "id", None) - if not call_id_raw: + call_id = "" + + if call_id_raw: + call_id = str(call_id_raw) + if tc_index is not None: + self._tool_call_id_by_index[tc_index] = call_id + elif tc_index is not None: + mapped_call_id = self._tool_call_id_by_index.get(tc_index) + if mapped_call_id: + call_id = mapped_call_id + + if not call_id: continue - call_id = str(call_id_raw) fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None) fn_name = "" diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py b/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py index 8d324bea611..4efdc217dce 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py @@ -229,3 +229,105 @@ def test_tool_call_arguments_are_chunked_to_match_openai_behavior(): assert sequence_numbers == sorted(sequence_numbers) assert len(set(sequence_numbers)) == len(sequence_numbers) # All unique + +def test_tool_call_delta_without_id_uses_index_mapping(): + iterator = LiteLLMCompletionStreamingIterator( + model="test-model", + litellm_custom_stream_wrapper=AsyncMock(), + request_input="Test input", + responses_api_request={}, + ) + + chunks = [ + [ + { + "index": 0, + "id": "call_abc123", + "type": "function", + "function": {"name": "get_weather", "arguments": '{"lo'}, + } + ], + [{"index": 0, "type": "function", "function": {"arguments": 'cation":'}}], + [{"index": 0, "type": "function", "function": {"arguments": ' "New'}}], + [{"index": 0, "type": "function", "function": {"arguments": ' York"}'}}], + ] + + for tool_calls in chunks: + iterator._queue_tool_call_delta_events(tool_calls) + + all_events = [] + while iterator._pending_tool_events: + all_events.append(iterator._pending_tool_events.pop(0)) + + delta_events = [ + evt + for evt in all_events + if evt.type == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA + ] + streamed_arguments = "".join(evt.delta for evt in delta_events) + + assert streamed_arguments == '{"location": "New York"}' + + output_item_added_events = [ + evt + for evt in all_events + if evt.type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED + ] + assert len(output_item_added_events) == 1 + assert output_item_added_events[0].item.id == "call_abc123" + + +def test_parallel_tool_calls_without_ids_use_index_mapping(): + iterator = LiteLLMCompletionStreamingIterator( + model="test-model", + litellm_custom_stream_wrapper=AsyncMock(), + request_input="Test input", + responses_api_request={}, + ) + + iterator._queue_tool_call_delta_events( + [ + { + "index": 0, + "id": "call_a", + "type": "function", + "function": {"name": "tool_a", "arguments": '{"x":'}, + }, + { + "index": 1, + "id": "call_b", + "type": "function", + "function": {"name": "tool_b", "arguments": '{"y":'}, + }, + ] + ) + iterator._queue_tool_call_delta_events( + [ + {"index": 0, "type": "function", "function": {"arguments": "1}"}}, + {"index": 1, "type": "function", "function": {"arguments": "2}"}}, + ] + ) + + all_events = [] + while iterator._pending_tool_events: + all_events.append(iterator._pending_tool_events.pop(0)) + + output_item_added_events = [ + evt + for evt in all_events + if evt.type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED + ] + assert len(output_item_added_events) == 2 + + delta_events = [ + evt + for evt in all_events + if evt.type == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA + ] + arguments_by_call_id = {} + for evt in delta_events: + arguments_by_call_id.setdefault(evt.item_id, "") + arguments_by_call_id[evt.item_id] += evt.delta + + assert arguments_by_call_id["call_a"] == '{"x":1}' + assert arguments_by_call_id["call_b"] == '{"y":2}' From cf17a440cdb4a8ca3055c91971ed2c1569b8c1c5 Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Sun, 8 Feb 2026 08:53:17 -0600 Subject: [PATCH 25/50] fix(responses): guard ambiguous tool-call index reuse --- .../streaming_iterator.py | 8 +++ ...test_tool_call_streaming_transformation.py | 59 +++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index ea9b8889d39..5c05526442d 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -89,6 +89,7 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self._tool_output_index_by_call_id: dict[str, int] = {} self._tool_args_by_call_id: dict[str, str] = {} self._tool_call_id_by_index: dict[int, str] = {} + self._ambiguous_tool_call_indexes: set[int] = set() self._next_tool_output_index: int = 1 # output_index=0 reserved for the message item self._final_tool_events_queued: bool = False self._sequence_number: int = 0 @@ -164,8 +165,15 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): if call_id_raw: call_id = str(call_id_raw) if tc_index is not None: + existing_call_id = self._tool_call_id_by_index.get(tc_index) + if existing_call_id is not None and existing_call_id != call_id: + # Reusing the same index for multiple call_ids is ambiguous for id-less deltas. + # Guard against silent misrouting by disabling index fallback for this index. + self._ambiguous_tool_call_indexes.add(tc_index) self._tool_call_id_by_index[tc_index] = call_id elif tc_index is not None: + if tc_index in self._ambiguous_tool_call_indexes: + continue mapped_call_id = self._tool_call_id_by_index.get(tc_index) if mapped_call_id: call_id = mapped_call_id diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py b/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py index 4efdc217dce..071eefaef47 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_tool_call_streaming_transformation.py @@ -331,3 +331,62 @@ def test_parallel_tool_calls_without_ids_use_index_mapping(): assert arguments_by_call_id["call_a"] == '{"x":1}' assert arguments_by_call_id["call_b"] == '{"y":2}' + + +def test_reused_index_with_new_call_id_marks_fallback_ambiguous(): + iterator = LiteLLMCompletionStreamingIterator( + model="test-model", + litellm_custom_stream_wrapper=AsyncMock(), + request_input="Test input", + responses_api_request={}, + ) + + iterator._queue_tool_call_delta_events( + [ + { + "index": 0, + "id": "call_a", + "type": "function", + "function": {"name": "tool_a", "arguments": '{"a":'}, + } + ] + ) + iterator._queue_tool_call_delta_events( + [ + { + "index": 0, + "id": "call_b", + "type": "function", + "function": {"name": "tool_b", "arguments": '{"b":'}, + } + ] + ) + # Ambiguous chunk: index reused and id missing. We should skip fallback rather than misroute. + iterator._queue_tool_call_delta_events( + [ + { + "index": 0, + "type": "function", + "function": {"arguments": "1}"}, + } + ] + ) + + all_events = [] + while iterator._pending_tool_events: + all_events.append(iterator._pending_tool_events.pop(0)) + + delta_events = [ + evt + for evt in all_events + if evt.type == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DELTA + ] + arguments_by_call_id = {} + for evt in delta_events: + arguments_by_call_id.setdefault(evt.item_id, "") + arguments_by_call_id[evt.item_id] += evt.delta + + assert arguments_by_call_id["call_a"] == '{"a":' + assert arguments_by_call_id["call_b"] == '{"b":' + assert arguments_by_call_id["call_a"] != '{"a":1}' + assert arguments_by_call_id["call_b"] != '{"b":1}' From 8cd8a01d5ac3946ac43032b14470319ec207bccc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 10:18:19 +0530 Subject: [PATCH 26/50] Add compaction for vertex ai --- .../anthropic_claude3_transformation.py | 29 ++--- .../transformation.py | 23 ++++ ...odel_prices_and_context_window_backup.json | 120 ------------------ model_prices_and_context_window.json | 120 ------------------ ...artner_models_anthropic_messages_config.py | 117 +++++++++++++++++ 5 files changed, 149 insertions(+), 260 deletions(-) diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index c936b2cd23c..31119c73d72 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -2,6 +2,7 @@ from typing import TYPE_CHECKING, Any, List, Optional import httpx +from litellm.anthropic_beta_headers_manager import filter_and_transform_beta_headers from litellm.llms.anthropic.chat.transformation import AnthropicConfig from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( AmazonInvokeConfig, @@ -133,27 +134,15 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): beta_set.add("tool-search-tool-2025-10-19") # Filter out beta headers that Bedrock Invoke doesn't support - # AWS Bedrock only supports a specific whitelist of beta flags - # Reference: https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-anthropic-claude-messages-request-response.html - BEDROCK_SUPPORTED_BETAS = { - "computer-use-2024-10-22", # Legacy computer use - "computer-use-2025-01-24", # Current computer use (Claude 3.7 Sonnet) - "token-efficient-tools-2025-02-19", # Tool use (Claude 3.7+ and Claude 4+) - "interleaved-thinking-2025-05-14", # Interleaved thinking (Claude 4+) - "output-128k-2025-02-19", # 128K output tokens (Claude 3.7 Sonnet) - "dev-full-thinking-2025-05-14", # Developer mode for raw thinking (Claude 4+) - "context-1m-2025-08-07", # 1 million tokens (Claude Sonnet 4) - "context-management-2025-06-27", # Context management (Claude Sonnet/Haiku 4.5) - "effort-2025-11-24", # Effort parameter (Claude Opus 4.5) - "tool-search-tool-2025-10-19", # Tool search (Claude Opus 4.5) - "tool-examples-2025-10-29", # Tool use examples (Claude Opus 4.5) - } - - # Only keep beta headers that Bedrock supports - beta_set = {beta for beta in beta_set if beta in BEDROCK_SUPPORTED_BETAS} + # Uses centralized configuration from anthropic_beta_headers_config.json + beta_list = list(beta_set) + filtered_beta_list = filter_and_transform_beta_headers( + beta_headers=beta_list, + provider="bedrock", + ) - if beta_set: - _anthropic_request["anthropic_beta"] = list(beta_set) + if filtered_beta_list: + _anthropic_request["anthropic_beta"] = filtered_beta_list return _anthropic_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 918b8ecc225..5a09168282d 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -68,6 +68,29 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert if existing_beta: beta_values.update(b.strip() for b in existing_beta.split(",")) + # Check for context management + context_management_param = optional_params.get("context_management") + if context_management_param is not None: + # Check edits array for compact_20260112 type + edits = context_management_param.get("edits", []) + has_compact = False + has_other = False + + for edit in edits: + edit_type = edit.get("type", "") + if edit_type == "compact_20260112": + has_compact = True + else: + has_other = True + + # Add compact header if any compact edits exist + if has_compact: + beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_01_12.value) + + # Add context management header if any other edits exist + if has_other: + beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.CONTEXT_MANAGEMENT_2025_06_27.value) + # Check for web search tool for tool in tools: if isinstance(tool, dict) and tool.get("type", "").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_SEARCH.value): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5cad0db241f..45dbf14ffc9 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -993,66 +993,6 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, - "anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, - "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, - "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, - "global.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, - "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, - "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, @@ -1143,66 +1083,6 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, - "eu.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, - "apac.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, "apac.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5cad0db241f..45dbf14ffc9 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -993,66 +993,6 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, - "anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, - "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, - "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, - "global.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, - "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, - "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, @@ -1143,66 +1083,6 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, - "eu.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, - "apac.anthropic.claude-opus-4-6-v1": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "litellm_provider": "bedrock_converse", - "max_input_tokens": 1000000, - "max_output_tokens": 128000, - "max_tokens": 128000, - "mode": "chat", - "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "search_context_cost_per_query": { - "search_context_size_high": 0.01, - "search_context_size_low": 0.01, - "search_context_size_medium": 0.01 - }, - "supports_assistant_prefill": false, - "supports_computer_use": true, - "supports_function_calling": true, - "supports_pdf_input": true, - "supports_prompt_caching": true, - "supports_reasoning": true, - "supports_response_schema": true, - "supports_tool_choice": true, - "supports_vision": true, - "tool_use_system_prompt_tokens": 346 - }, "apac.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py index 623f8c579ff..7bb84b0a2c1 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py @@ -98,3 +98,120 @@ def test_web_search_header_not_added_without_tool(): # Assert that the anthropic-beta header is NOT present when no web search tool assert "anthropic-beta" not in updated_headers, \ "anthropic-beta header should not be present without web search tool" + + +def test_compact_context_management_header_added(): + """Test that compact-2026-01-12 beta header is added when context_management with compact_20260112 is used""" + config = VertexAIPartnerModelsAnthropicMessagesConfig() + headers = {} + litellm_params = { + "vertex_ai_project": "test-project", + "vertex_ai_location": "us-central1", + "vertex_credentials": "{}", + } + # Include context_management with compact_20260112 + optional_params = { + "context_management": { + "edits": [ + {"type": "compact_20260112"} + ] + } + } + + with patch.object( + config, "_ensure_access_token", return_value=("token", "test-project") + ), patch.object( + config, "get_complete_vertex_url", return_value="https://mock-url" + ): + updated_headers, api_base = config.validate_anthropic_messages_environment( + headers=headers, + model="claude-vertex-ai-opus-4-6", + messages=[], + optional_params=optional_params, + litellm_params=litellm_params, + api_base=None, + ) + + # Assert that the anthropic-beta header with compact-2026-01-12 is present + assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present" + assert "compact-2026-01-12" in updated_headers["anthropic-beta"], \ + f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}" + + +def test_context_management_header_added_for_other_edits(): + """Test that context-management-2025-06-27 beta header is added for non-compact edits""" + config = VertexAIPartnerModelsAnthropicMessagesConfig() + headers = {} + litellm_params = { + "vertex_ai_project": "test-project", + "vertex_ai_location": "us-central1", + "vertex_credentials": "{}", + } + # Include context_management with other edit types + optional_params = { + "context_management": { + "edits": [ + {"type": "some_other_type"} + ] + } + } + + with patch.object( + config, "_ensure_access_token", return_value=("token", "test-project") + ), patch.object( + config, "get_complete_vertex_url", return_value="https://mock-url" + ): + updated_headers, api_base = config.validate_anthropic_messages_environment( + headers=headers, + model="claude-vertex-ai-opus-4-6", + messages=[], + optional_params=optional_params, + litellm_params=litellm_params, + api_base=None, + ) + + # Assert that the anthropic-beta header with context-management-2025-06-27 is present + assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present" + assert "context-management-2025-06-27" in updated_headers["anthropic-beta"], \ + f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}" + + +def test_both_compact_and_context_management_headers_added(): + """Test that both compact and context-management beta headers are added when both edit types are present""" + config = VertexAIPartnerModelsAnthropicMessagesConfig() + headers = {} + litellm_params = { + "vertex_ai_project": "test-project", + "vertex_ai_location": "us-central1", + "vertex_credentials": "{}", + } + # Include context_management with both compact and other edit types + optional_params = { + "context_management": { + "edits": [ + {"type": "compact_20260112"}, + {"type": "some_other_type"} + ] + } + } + + with patch.object( + config, "_ensure_access_token", return_value=("token", "test-project") + ), patch.object( + config, "get_complete_vertex_url", return_value="https://mock-url" + ): + updated_headers, api_base = config.validate_anthropic_messages_environment( + headers=headers, + model="claude-vertex-ai-opus-4-6", + messages=[], + optional_params=optional_params, + litellm_params=litellm_params, + api_base=None, + ) + + # Assert that both beta headers are present + assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present" + assert "compact-2026-01-12" in updated_headers["anthropic-beta"], \ + f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}" + assert "context-management-2025-06-27" in updated_headers["anthropic-beta"], \ + f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}" From d41df6053a0ba8346db4cc0f0f87ad180442e148 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 10:35:11 +0530 Subject: [PATCH 27/50] Add all new feat for v1/messages --- docs/my-website/blog/claude_opus_4_6/index.md | 127 ++++++++++++++++++ .../anthropic/messages_transformation.py | 9 ++ 2 files changed, 136 insertions(+) diff --git a/docs/my-website/blog/claude_opus_4_6/index.md b/docs/my-website/blog/claude_opus_4_6/index.md index 0397f1288f7..47dcd629226 100644 --- a/docs/my-website/blog/claude_opus_4_6/index.md +++ b/docs/my-website/blog/claude_opus_4_6/index.md @@ -401,3 +401,130 @@ Opus 4.6 supports 1M token context. Premium pricing applies for prompts exceedin Available at 1.1Γ— token pricing. LiteLLM supports this pricing model. +## Using `/v1/messages` Endpoint + +LiteLLM supports the Anthropic `/v1/messages` API format across all providers. This allows you to use Anthropic-specific features like adaptive thinking, compaction, and 1M token context with consistent syntax across Anthropic, Azure AI, Vertex AI, and Bedrock. + + +```yaml +model_list: + # Anthropic + - model_name: claude-opus-4-6 + litellm_params: + model: anthropic/claude-opus-4-6 + + # Azure AI + - model_name: claude-azure-opus-4-6 + litellm_params: + model: azure_ai/claude-opus-4-6 + api_base: https://your-resource.services.ai.azure.com/anthropic + + # Vertex AI + - model_name: claude-vertex-opus-4-6 + litellm_params: + model: vertex_ai/claude-opus-4-6 + vertex_project: your-project-id + vertex_location: us-east5 + + # Bedrock + - model_name: claude-bedrock-opus-4-6 + litellm_params: + model: bedrock/anthropic.claude-opus-4-6-v1:0 + aws_region_name: us-east-1 +``` + +### Feature Support Matrix + +| Feature | Anthropic | Bedrock Invoke | Bedrock Converse | Vertex AI | Azure AI | +|---------|-----------|----------------|------------------|-----------|----------| +| `/v1/messages` Compaction | βœ… | ❌ Not supported | ❌ Not supported | βœ… | βœ… | +| 1M Token Context | βœ… | βœ… | βœ… | βœ… | βœ… | +| US-Only Inference - cost tracking | βœ… | Not applicable | Not applicable | Not applicable | Not applicable | +| Adaptive Thinking | βœ… | βœ… | βœ… | βœ… | βœ… | + +### Adaptive Thinking + +Use the `thinking` parameter with `type: "adaptive"` to enable adaptive thinking mode: + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 16000, + "thinking": { + "type": "adaptive" + }, + "messages": [ + { + "role": "user", + "content": "Explain why the sum of two even numbers is always even." + } + ] +}' +``` + +### Context Management - Compaction + +Enable compaction to reduce context size while preserving key information. LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled. + +:::info +**Provider Support:** Compaction is supported on Anthropic, Bedrock Invoke Azure AI, and Vertex AI. It is **not supported** on Bedrock Converse API. +::: + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 4096, + "messages": [ + { + "role": "user", + "content": "Hi" + } + ], + "context_management": { + "edits": [ + { + "type": "compact_20260112" + } + ] + } +}' +``` + +LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled. + +### 1M Token Context Window + +To use the 1M token context window, you need to forward the `anthropic-beta` header from your client to the LLM provider. + +**Step 1: Enable header forwarding in your config** + +```yaml +general_settings: + forward_client_headers_to_llm_api: true +``` + +**Step 2: Send requests with the beta header** + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'anthropic-beta: context-1m-2025-08-07' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "Explain why the sum of two even numbers is always even." + } + ] +}' +``` + diff --git a/litellm/llms/azure_ai/anthropic/messages_transformation.py b/litellm/llms/azure_ai/anthropic/messages_transformation.py index 0d00c907031..f86ec7082f2 100644 --- a/litellm/llms/azure_ai/anthropic/messages_transformation.py +++ b/litellm/llms/azure_ai/anthropic/messages_transformation.py @@ -3,6 +3,9 @@ Azure Anthropic messages transformation config - extends AnthropicMessagesConfig """ from typing import TYPE_CHECKING, Any, List, Optional, Tuple +from litellm.anthropic_beta_headers_manager import ( + update_headers_with_filtered_beta, +) from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) @@ -68,6 +71,12 @@ class AzureAnthropicMessagesConfig(AnthropicMessagesConfig): optional_params=optional_params, ) + # Filter out unsupported beta headers for Azure AI + headers = update_headers_with_filtered_beta( + headers=headers, + provider="azure_ai", + ) + return headers, api_base def get_complete_url( From 3307f3d1c6c069ff0b9505bfdd4879f84771ea56 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 10:47:13 +0530 Subject: [PATCH 28/50] Add inference_geo as supported messages param --- .../experimental_pass_through/messages/transformation.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index bb40f9df266..145a7157139 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -46,6 +46,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "thinking", "context_management", "output_format", + "inference_geo", # TODO: Add Anthropic `metadata` support # "metadata", ] From 20440bcadca46aec3aaa5704e4afc5459902251f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 10:51:50 +0530 Subject: [PATCH 29/50] Add inference based costing --- docs/my-website/blog/claude_opus_4_6/index.md | 283 ++++++++++++------ 1 file changed, 187 insertions(+), 96 deletions(-) diff --git a/docs/my-website/blog/claude_opus_4_6/index.md b/docs/my-website/blog/claude_opus_4_6/index.md index 47dcd629226..78411b90ba7 100644 --- a/docs/my-website/blog/claude_opus_4_6/index.md +++ b/docs/my-website/blog/claude_opus_4_6/index.md @@ -223,11 +223,16 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ -## Compaction +## Advanced Features + +### Compaction + + + Litellm supports enabling compaction for the new claude-opus-4-6. -### Enabling Compaction +**Enabling Compaction** To enable compaction, add the `context_management` parameter with the `compact_20260112` edit type: @@ -255,8 +260,43 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ ``` All the parameters supported for context_management by anthropic are supported and can be directly added. Litellm automatically adds the `compact-2026-01-12` beta header in the request. + + -### Response with Compaction Block +Enable compaction to reduce context size while preserving key information. LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled. + +:::info +**Provider Support:** Compaction is supported on Anthropic, Azure AI, and Vertex AI. It is **not supported** on Bedrock (Invoke or Converse APIs). +::: + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 4096, + "messages": [ + { + "role": "user", + "content": "Hi" + } + ], + "context_management": { + "edits": [ + { + "type": "compact_20260112" + } + ] + } +}' +``` + + + + + +**Response with Compaction Block** The response will include the compaction summary in `provider_specific_fields.compaction_blocks`: @@ -292,7 +332,7 @@ The response will include the compaction summary in `provider_specific_fields.co } ``` -### Using Compaction Blocks in Follow-up Requests +**Using Compaction Blocks in Follow-up Requests** To continue the conversation with compaction, include the compaction block in the assistant message's `provider_specific_fields`: @@ -340,15 +380,17 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ }' ``` -### Streaming Support +**Streaming Support** Compaction blocks are also supported in streaming mode. You'll receive: - `compaction_start` event when a compaction block begins - `compaction_delta` events with the compaction content - The accumulated `compaction_blocks` in `provider_specific_fields` +### Adaptive Thinking -## Adaptive Thinking + + LiteLLM supports adaptive thinking through the `reasoning_effort` parameter: @@ -368,81 +410,8 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ }' ``` -## Effort Levels - -Four effort levels available: `low`, `medium`, `high` (default), and `max`. Pass directly via the `output_config` parameter: - -```bash -curl --location 'http://0.0.0.0:4000/chat/completions' \ ---header 'Content-Type: application/json' \ ---header 'Authorization: Bearer $LITELLM_KEY' \ ---data '{ - "model": "claude-opus-4-6", - "messages": [ - { - "role": "user", - "content": "Explain quantum computing" - } - ], - "output_config": { - "effort": "medium" - } - -}' -``` - -You can use reasoning effort plus output_config to have more control on the model. - -## 1M Token Context (Beta) - -Opus 4.6 supports 1M token context. Premium pricing applies for prompts exceeding 200k tokens ($10/$37.50 per million input/output tokens). LiteLLM supports cost calculations for 1M token contexts. - -## US-Only Inference - -Available at 1.1Γ— token pricing. LiteLLM supports this pricing model. - -## Using `/v1/messages` Endpoint - -LiteLLM supports the Anthropic `/v1/messages` API format across all providers. This allows you to use Anthropic-specific features like adaptive thinking, compaction, and 1M token context with consistent syntax across Anthropic, Azure AI, Vertex AI, and Bedrock. - - -```yaml -model_list: - # Anthropic - - model_name: claude-opus-4-6 - litellm_params: - model: anthropic/claude-opus-4-6 - - # Azure AI - - model_name: claude-azure-opus-4-6 - litellm_params: - model: azure_ai/claude-opus-4-6 - api_base: https://your-resource.services.ai.azure.com/anthropic - - # Vertex AI - - model_name: claude-vertex-opus-4-6 - litellm_params: - model: vertex_ai/claude-opus-4-6 - vertex_project: your-project-id - vertex_location: us-east5 - - # Bedrock - - model_name: claude-bedrock-opus-4-6 - litellm_params: - model: bedrock/anthropic.claude-opus-4-6-v1:0 - aws_region_name: us-east-1 -``` - -### Feature Support Matrix - -| Feature | Anthropic | Bedrock Invoke | Bedrock Converse | Vertex AI | Azure AI | -|---------|-----------|----------------|------------------|-----------|----------| -| `/v1/messages` Compaction | βœ… | ❌ Not supported | ❌ Not supported | βœ… | βœ… | -| 1M Token Context | βœ… | βœ… | βœ… | βœ… | βœ… | -| US-Only Inference - cost tracking | βœ… | Not applicable | Not applicable | Not applicable | Not applicable | -| Adaptive Thinking | βœ… | βœ… | βœ… | βœ… | βœ… | - -### Adaptive Thinking + + Use the `thinking` parameter with `type: "adaptive"` to enable adaptive thinking mode: @@ -465,13 +434,40 @@ curl --location 'http://0.0.0.0:4000/v1/messages' \ }' ``` -### Context Management - Compaction + + -Enable compaction to reduce context size while preserving key information. LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled. +### Effort Levels -:::info -**Provider Support:** Compaction is supported on Anthropic, Bedrock Invoke Azure AI, and Vertex AI. It is **not supported** on Bedrock Converse API. -::: + + + +Four effort levels available: `low`, `medium`, `high` (default), and `max`. Pass directly via the `output_config` parameter: + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer $LITELLM_KEY' \ +--data '{ + "model": "claude-opus-4-6", + "messages": [ + { + "role": "user", + "content": "Explain quantum computing" + } + ], + "output_config": { + "effort": "medium" + } +}' +``` + +You can use reasoning effort plus output_config to have more control on the model. + + + + +Four effort levels available: `low`, `medium`, `high` (default), and `max`. Pass directly via the `output_config` parameter: ```bash curl --location 'http://0.0.0.0:4000/v1/messages' \ @@ -483,22 +479,54 @@ curl --location 'http://0.0.0.0:4000/v1/messages' \ "messages": [ { "role": "user", - "content": "Hi" + "content": "Explain quantum computing" } ], - "context_management": { - "edits": [ - { - "type": "compact_20260112" - } - ] + "output_config": { + "effort": "medium" } }' ``` -LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled. + + -### 1M Token Context Window +### 1M Token Context (Beta) + +Opus 4.6 supports 1M token context. Premium pricing applies for prompts exceeding 200k tokens ($10/$37.50 per million input/output tokens). LiteLLM supports cost calculations for 1M token contexts. + + + + +To use the 1M token context window, you need to forward the `anthropic-beta` header from your client to the LLM provider. + +**Step 1: Enable header forwarding in your config** + +```yaml +general_settings: + forward_client_headers_to_llm_api: true +``` + +**Step 2: Send requests with the beta header** + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer $LITELLM_KEY' \ +--header 'anthropic-beta: context-1m-2025-08-07' \ +--data '{ + "model": "claude-opus-4-6", + "messages": [ + { + "role": "user", + "content": "Analyze this large document..." + } + ] +}' +``` + + + To use the 1M token context window, you need to forward the `anthropic-beta` header from your client to the LLM provider. @@ -522,9 +550,72 @@ curl --location 'http://0.0.0.0:4000/v1/messages' \ "messages": [ { "role": "user", - "content": "Explain why the sum of two even numbers is always even." + "content": "Analyze this large document..." } ] }' ``` +:::tip +You can combine multiple beta headers by separating them with commas: +```bash +--header 'anthropic-beta: context-1m-2025-08-07,compact-2026-01-12' +``` +::: + + + + +### US-Only Inference + +Available at 1.1Γ— token pricing. LiteLLM automatically tracks costs for US-only inference. + + + + +Use the `inference_geo` parameter to specify US-only inference: + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer $LITELLM_KEY' \ +--data '{ + "model": "claude-opus-4-6", + "messages": [ + { + "role": "user", + "content": "What is the capital of France?" + } + ], + "inference_geo": "us" +}' +``` + +LiteLLM will automatically apply the 1.1Γ— pricing multiplier for US-only inference in cost tracking. + + + + +Use the `inference_geo` parameter to specify US-only inference: + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 4096, + "messages": [ + { + "role": "user", + "content": "What is the capital of France?" + } + ], + "inference_geo": "us" +}' +``` + +LiteLLM will automatically apply the 1.1Γ— pricing multiplier for US-only inference in cost tracking. + + + From 29e6efade9efe45ef88d2cc2ac334dda8ed2d190 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 10:57:04 +0530 Subject: [PATCH 30/50] Add inference_geo as supported messages param --- litellm/types/llms/anthropic.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index fedf419efd6..d5172c7a266 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -355,6 +355,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False): tool_choice: Optional[Union[AnthropicMessagesToolChoice, Dict]] tools: Optional[List[Union[AllAnthropicToolsValues, Dict]]] top_k: Optional[int] + inference_geo: Optional[str] top_p: Optional[float] mcp_servers: Optional[List[AnthropicMcpServerTool]] context_management: Optional[Dict[str, Any]] From b822e2e0ffe58ee2bbaae91038c7fa0f6ed33910 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 11:28:00 +0530 Subject: [PATCH 31/50] Add support for fast param --- litellm/llms/anthropic/chat/handler.py | 12 +- litellm/llms/anthropic/chat/transformation.py | 14 ++ litellm/llms/anthropic/cost_calculation.py | 15 +- .../messages/transformation.py | 8 +- ...odel_prices_and_context_window_backup.json | 93 ++++++++++ litellm/types/llms/anthropic.py | 2 + model_prices_and_context_window.json | 93 ++++++++++ .../test_anthropic_chat_transformation.py | 161 ++++++++++++++++++ 8 files changed, 389 insertions(+), 9 deletions(-) diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index 485e95d6489..e85c0d0d017 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -75,6 +75,7 @@ async def make_call( logging_obj, timeout: Optional[Union[float, httpx.Timeout]], json_mode: bool, + speed: Optional[str] = None, ) -> Tuple[Any, httpx.Headers]: if client is None: client = litellm.module_level_aclient @@ -103,6 +104,7 @@ async def make_call( streaming_response=response.aiter_lines(), sync_stream=False, json_mode=json_mode, + speed=speed, ) # LOGGING @@ -126,6 +128,7 @@ def make_sync_call( logging_obj, timeout: Optional[Union[float, httpx.Timeout]], json_mode: bool, + speed: Optional[str] = None, ) -> Tuple[Any, httpx.Headers]: if client is None: client = litellm.module_level_client # re-use a module level client @@ -159,7 +162,7 @@ def make_sync_call( ) completion_stream = ModelResponseIterator( - streaming_response=response.iter_lines(), sync_stream=True, json_mode=json_mode + streaming_response=response.iter_lines(), sync_stream=True, json_mode=json_mode, speed=speed ) # LOGGING @@ -213,6 +216,7 @@ class AnthropicChatCompletion(BaseLLM): logging_obj=logging_obj, timeout=timeout, json_mode=json_mode, + speed=optional_params.get("speed") if optional_params else None, ) streamwrapper = CustomStreamWrapper( completion_stream=completion_stream, @@ -427,6 +431,7 @@ class AnthropicChatCompletion(BaseLLM): logging_obj=logging_obj, timeout=timeout, json_mode=json_mode, + speed=optional_params.get("speed") if optional_params else None, ) return CustomStreamWrapper( completion_stream=completion_stream, @@ -485,13 +490,14 @@ class AnthropicChatCompletion(BaseLLM): class ModelResponseIterator: def __init__( - self, streaming_response, sync_stream: bool, json_mode: Optional[bool] = False + self, streaming_response, sync_stream: bool, json_mode: Optional[bool] = False, speed: Optional[str] = None ): self.streaming_response = streaming_response self.response_iterator = self.streaming_response self.content_blocks: List[ContentBlockDelta] = [] self.tool_index = -1 self.json_mode = json_mode + self.speed = speed # Generate response ID once per stream to match OpenAI-compatible behavior self.response_id = _generate_id() @@ -541,7 +547,7 @@ class ModelResponseIterator: def _handle_usage(self, anthropic_usage_chunk: Union[dict, UsageDelta]) -> Usage: return AnthropicConfig().calculate_usage( - usage_object=cast(dict, anthropic_usage_chunk), reasoning_content=None + usage_object=cast(dict, anthropic_usage_chunk), reasoning_content=None, speed=self.speed ) def _content_block_delta_helper(self, chunk: dict) -> Tuple[ diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 02b8d952445..82aa7390188 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -190,6 +190,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): "response_format", "user", "web_search_options", + "speed", ] if "claude-3-7-sonnet" in model or supports_reasoning( @@ -882,6 +883,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): elif param == "context_management" and isinstance(value, dict): # Pass through Anthropic-specific context_management parameter optional_params["context_management"] = value + elif param == "speed" and isinstance(value, str): + # Pass through Anthropic-specific speed parameter for fast mode + optional_params["speed"] = value ## handle thinking tokens self.update_optional_params_with_thinking_tokens( @@ -1096,6 +1100,10 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): self._ensure_beta_header( headers, ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value ) + if optional_params.get("speed") == "fast": + self._ensure_beta_header( + headers, ANTHROPIC_BETA_HEADER_VALUES.FAST_MODE_2026_02_01.value + ) return headers def transform_request( @@ -1349,6 +1357,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): usage_object: dict, reasoning_content: Optional[str], completion_response: Optional[dict] = None, + speed: Optional[str] = None, ) -> Usage: # NOTE: Sometimes the usage object has None set explicitly for token counts, meaning .get() & key access returns None, and we need to account for this prompt_tokens = usage_object.get("input_tokens", 0) or 0 @@ -1447,6 +1456,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): else None ), inference_geo=inference_geo, + speed=speed, ) return usage @@ -1457,6 +1467,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response: ModelResponse, json_mode: Optional[bool] = None, prefix_prompt: Optional[str] = None, + speed: Optional[str] = None, ): _hidden_params: Dict = {} _hidden_params["additional_headers"] = process_anthropic_headers( @@ -1553,6 +1564,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): usage_object=completion_response["usage"], reasoning_content=reasoning_content, completion_response=completion_response, + speed=speed, ) setattr(model_response, "usage", usage) # type: ignore @@ -1621,6 +1633,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) prefix_prompt = self.get_prefix_prompt(messages=messages) + speed = optional_params.get("speed") model_response = self.transform_parsed_response( completion_response=completion_response, @@ -1628,6 +1641,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response=model_response, json_mode=json_mode, prefix_prompt=prefix_prompt, + speed=speed, ) return model_response diff --git a/litellm/llms/anthropic/cost_calculation.py b/litellm/llms/anthropic/cost_calculation.py index 11b61cc92f0..271406f2f7d 100644 --- a/litellm/llms/anthropic/cost_calculation.py +++ b/litellm/llms/anthropic/cost_calculation.py @@ -22,13 +22,18 @@ def cost_per_token(model: str, usage: "Usage") -> Tuple[float, float]: Returns: Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd """ - # If usage has inference_geo, prepend it as prefix to model name + model_with_prefix = model + + # First, prepend inference_geo if present if hasattr(usage, "inference_geo") and usage.inference_geo and usage.inference_geo.lower() not in ["global", "not_available"]: - model_with_geo_prefix = f"{usage.inference_geo}/{model}" - else: - model_with_geo_prefix = model + model_with_prefix = f"{usage.inference_geo}/{model_with_prefix}" + + # Then, prepend speed if it's "fast" + if hasattr(usage, "speed") and usage.speed == "fast": + model_with_prefix = f"fast/{model_with_prefix}" + prompt_cost, completion_cost = generic_cost_per_token( - model=model_with_geo_prefix, usage=usage, custom_llm_provider="anthropic" + model=model_with_prefix, usage=usage, custom_llm_provider="anthropic" ) return prompt_cost, completion_cost diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index 145a7157139..9e28c139686 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -47,6 +47,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "context_management", "output_format", "inference_geo", + "speed", # TODO: Add Anthropic `metadata` support # "metadata", ] @@ -184,10 +185,11 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): - context_management: adds 'context-management-2025-06-27' - tool_search: adds provider-specific tool search header - output_format: adds 'structured-outputs-2025-11-13' + - speed: adds 'fast-mode-2026-02-01' Args: headers: Request headers dict - optional_params: Optional parameters including tools, context_management, output_format + optional_params: Optional parameters including tools, context_management, output_format, speed custom_llm_provider: Provider name for looking up correct tool search header """ beta_values: set = set() @@ -224,6 +226,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): if optional_params.get("output_format") is not None: beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.STRUCTURED_OUTPUT_2025_09_25.value) + # Check for fast mode + if optional_params.get("speed") == "fast": + beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.FAST_MODE_2026_02_01.value) + # Check for tool search tools tools = optional_params.get("tools") if tools: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 45dbf14ffc9..543f2f14d7a 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -7663,6 +7663,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/claude-opus-4-6": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 3.75e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "us/claude-opus-4-6": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, @@ -7694,6 +7725,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/us/claude-opus-4-6": { + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1.1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 4.125e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, @@ -7725,6 +7787,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/claude-opus-4-6-20260205": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 3.75e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "us/claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index d5172c7a266..84ac01a4ece 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -361,6 +361,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False): context_management: Optional[Dict[str, Any]] container: Optional[Dict[str, Any]] # Container config with skills for code execution output_format: Optional[AnthropicOutputSchema] # Structured outputs support + speed: Optional[str] # Fast mode support for Opus models class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False): @@ -637,6 +638,7 @@ class ANTHROPIC_BETA_HEADER_VALUES(str, Enum): COMPACT_2026_01_12 = "compact-2026-01-12" STRUCTURED_OUTPUT_2025_09_25 = "structured-outputs-2025-11-13" ADVANCED_TOOL_USE_2025_11_20 = "advanced-tool-use-2025-11-20" + FAST_MODE_2026_02_01 = "fast-mode-2026-02-01" # Tool search beta header constant (for Anthropic direct API and Microsoft Foundry) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 45dbf14ffc9..543f2f14d7a 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -7663,6 +7663,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/claude-opus-4-6": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 3.75e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "us/claude-opus-4-6": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, @@ -7694,6 +7725,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/us/claude-opus-4-6": { + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1.1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 200000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 4.125e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.25e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, @@ -7725,6 +7787,37 @@ "supports_vision": true, "tool_use_system_prompt_tokens": 346 }, + "fast/claude-opus-4-6-20260205": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_200k_tokens": 1e-06, + "input_cost_per_token": 3e-05, + "input_cost_per_token_above_200k_tokens": 1e-05, + "litellm_provider": "anthropic", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 0.00015, + "output_cost_per_token_above_200k_tokens": 3.75e-05, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 346 + }, "us/claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.875e-06, "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 49db7367c67..e3bd7d2bb31 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -2506,3 +2506,164 @@ def test_compaction_block_empty_list_not_added(): provider_fields = result.choices[0].message.provider_specific_fields if provider_fields: assert "compaction_blocks" not in provider_fields or provider_fields.get("compaction_blocks") is None + + +def test_fast_mode_beta_header(): + """ + Test that fast mode correctly adds the fast-mode-2026-02-01 beta header. + """ + config = AnthropicConfig() + + headers = {} + optional_params = {"speed": "fast"} + + result_headers = config.update_headers_with_optional_anthropic_beta( + headers=headers, + optional_params=optional_params + ) + + assert "anthropic-beta" in result_headers + assert "fast-mode-2026-02-01" in result_headers["anthropic-beta"] + + +def test_fast_mode_with_other_beta_headers(): + """ + Test that fast mode beta header is combined with other beta headers. + """ + config = AnthropicConfig() + + headers = {} + optional_params = { + "speed": "fast", + "output_format": {"type": "json_object"} + } + + result_headers = config.update_headers_with_optional_anthropic_beta( + headers=headers, + optional_params=optional_params + ) + + assert "anthropic-beta" in result_headers + assert "fast-mode-2026-02-01" in result_headers["anthropic-beta"] + assert "structured-outputs-2025-11-13" in result_headers["anthropic-beta"] + + +def test_fast_mode_usage_calculation(): + """ + Test that fast mode speed parameter is passed through to usage object. + """ + config = AnthropicConfig() + + usage_object = { + "input_tokens": 1000, + "output_tokens": 500, + } + + usage = config.calculate_usage( + usage_object=usage_object, + reasoning_content=None, + speed="fast" + ) + + assert usage.prompt_tokens == 1000 + assert usage.completion_tokens == 500 + assert hasattr(usage, "speed") + assert usage.speed == "fast" + + +def test_fast_mode_cost_calculation(): + """ + Test that fast mode correctly prepends 'fast/' to model name for pricing lookup. + """ + from unittest.mock import patch + + from litellm.llms.anthropic.cost_calculation import cost_per_token + from litellm.types.utils import Usage + + # Mock the generic_cost_per_token to verify correct model name is passed + with patch('litellm.llms.anthropic.cost_calculation.generic_cost_per_token') as mock_cost: + mock_cost.return_value = (0.03, 0.15) # $30 and $150 per MTok + + # Test fast mode + usage_fast = Usage( + prompt_tokens=1000, + completion_tokens=1000, + speed="fast" + ) + + prompt_cost, completion_cost = cost_per_token( + model="claude-opus-4-6", + usage=usage_fast + ) + + # Verify that generic_cost_per_token was called with "fast/claude-opus-4-6" + mock_cost.assert_called_once() + call_args = mock_cost.call_args + assert call_args[1]['model'] == "fast/claude-opus-4-6" + assert call_args[1]['custom_llm_provider'] == "anthropic" + + +def test_fast_mode_with_inference_geo(): + """ + Test that fast mode works correctly with inference_geo prefix. + Expected format: fast/us/claude-opus-4-6 + """ + from unittest.mock import patch + + from litellm.llms.anthropic.cost_calculation import cost_per_token + from litellm.types.utils import Usage + + # Mock the generic_cost_per_token to verify correct model name is passed + with patch('litellm.llms.anthropic.cost_calculation.generic_cost_per_token') as mock_cost: + mock_cost.return_value = (0.03, 0.15) + + # Test with both speed and inference_geo + usage = Usage( + prompt_tokens=1000, + completion_tokens=1000, + speed="fast", + inference_geo="us" + ) + + # This should look up "fast/us/claude-opus-4-6" in pricing + prompt_cost, completion_cost = cost_per_token( + model="claude-opus-4-6", + usage=usage + ) + + # Verify that generic_cost_per_token was called with "fast/us/claude-opus-4-6" + mock_cost.assert_called_once() + call_args = mock_cost.call_args + assert call_args[1]['model'] == "fast/us/claude-opus-4-6" + assert call_args[1]['custom_llm_provider'] == "anthropic" + + +def test_fast_mode_parameter_in_supported_params(): + """ + Test that 'speed' is in the list of supported OpenAI params. + """ + config = AnthropicConfig() + + supported_params = config.get_supported_openai_params(model="claude-opus-4-6") + + assert "speed" in supported_params + + +def test_fast_mode_parameter_mapping(): + """ + Test that speed parameter is correctly mapped in map_openai_params. + """ + config = AnthropicConfig() + + non_default_params = {"speed": "fast"} + optional_params = {} + + result = config.map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model="claude-opus-4-6", + drop_params=False + ) + + assert "speed" in result + assert result["speed"] == "fast" From c3b1c0a59068ec618e9f3f3af6faaf5c3b7c2754 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 11:36:10 +0530 Subject: [PATCH 32/50] Add fast mode for other providers --- litellm/anthropic_beta_headers_config.json | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/anthropic_beta_headers_config.json b/litellm/anthropic_beta_headers_config.json index 193091c0176..1e99e35aabe 100644 --- a/litellm/anthropic_beta_headers_config.json +++ b/litellm/anthropic_beta_headers_config.json @@ -13,7 +13,8 @@ "web-fetch-2025-09-10", "code-execution-2025-08-25", "skills-2025-10-02", - "files-api-2025-04-14" + "files-api-2025-04-14", + "fast-mode-2026-02-01" ], "bedrock": [ "advanced-tool-use-2025-11-20", @@ -22,7 +23,8 @@ "web-fetch-2025-09-10", "code-execution-2025-08-25", "skills-2025-10-02", - "files-api-2025-04-14" + "files-api-2025-04-14", + "fast-mode-2026-02-01" ], "vertex_ai": [ "prompt-caching-scope-2026-01-05" From 319453d059e69e46117f34d309a5a0acb19bdcb3 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 11:39:35 +0530 Subject: [PATCH 33/50] Add documentation for Fast Mode --- docs/my-website/blog/claude_opus_4_6/index.md | 90 +++++++++++++++++++ 1 file changed, 90 insertions(+) diff --git a/docs/my-website/blog/claude_opus_4_6/index.md b/docs/my-website/blog/claude_opus_4_6/index.md index 78411b90ba7..3fd70661543 100644 --- a/docs/my-website/blog/claude_opus_4_6/index.md +++ b/docs/my-website/blog/claude_opus_4_6/index.md @@ -619,3 +619,93 @@ LiteLLM will automatically apply the 1.1Γ— pricing multiplier for US-only infere + +### Fast Mode + +:::info +Fast mode is **only supported on the Anthropic provider** (`anthropic/claude-opus-4-6`). It is not available on Azure AI, Vertex AI, or Bedrock. +::: + +**Pricing:** +- Standard: $5 input / $25 output per MTok +- Fast: $30 input / $150 output per MTok (6Γ— premium) + + + + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer $LITELLM_KEY' \ +--data '{ + "model": "claude-opus-4-6", + "messages": [ + { + "role": "user", + "content": "Refactor this module..." + } + ], + "max_tokens": 4096, + "speed": "fast" +}' +``` + +**Using OpenAI SDK:** + +```python +import openai + +client = openai.OpenAI( + api_key="your-litellm-key", + base_url="http://0.0.0.0:4000" +) + +response = client.chat.completions.create( + model="claude-opus-4-6", + messages=[{"role": "user", "content": "Refactor this module..."}], + max_tokens=4096, + extra_body={"speed": "fast"} +) +``` + +**Using LiteLLM SDK:** + +```python +from litellm import completion + +response = completion( + model="anthropic/claude-opus-4-6", + messages=[{"role": "user", "content": "Refactor this module..."}], + max_tokens=4096, + speed="fast" +) +``` + +LiteLLM automatically tracks the higher costs for fast mode in usage and cost calculations. + + + + +```bash +curl --location 'http://0.0.0.0:4000/v1/messages' \ +--header 'x-api-key: sk-12345' \ +--header 'content-type: application/json' \ +--data '{ + "model": "claude-opus-4-6", + "max_tokens": 4096, + "speed": "fast", + "messages": [ + { + "role": "user", + "content": "Refactor this module..." + } + ] +}' +``` + +LiteLLM automatically: +- Adds the `fast-mode-2026-02-01` beta header +- Tracks the 6Γ— premium pricing in cost calculations + + + From 248fe6573635078a4f59196cab0afa377ae57719 Mon Sep 17 00:00:00 2001 From: Carlo Alberto Ferraris Date: Fri, 30 Jan 2026 12:01:14 +0900 Subject: [PATCH 34/50] add missing indexes on VerificationToken table --- .../migration.sql | 8 ++++++++ .../litellm_proxy_extras/schema.prisma | 10 ++++++++++ litellm/proxy/schema.prisma | 10 ++++++++++ schema.prisma | 10 ++++++++++ 4 files changed, 38 insertions(+) create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260209085821_add_verificationtoken_indexes/migration.sql diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260209085821_add_verificationtoken_indexes/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260209085821_add_verificationtoken_indexes/migration.sql new file mode 100644 index 00000000000..572eea9b529 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260209085821_add_verificationtoken_indexes/migration.sql @@ -0,0 +1,8 @@ +-- CreateIndex +CREATE INDEX "LiteLLM_VerificationToken_user_id_team_id_idx" ON "LiteLLM_VerificationToken"("user_id", "team_id"); + +-- CreateIndex +CREATE INDEX "LiteLLM_VerificationToken_team_id_idx" ON "LiteLLM_VerificationToken"("team_id"); + +-- CreateIndex +CREATE INDEX "LiteLLM_VerificationToken_budget_reset_at_expires_idx" ON "LiteLLM_VerificationToken"("budget_reset_at", "expires"); diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index c2a599c178f..b1ca1f71c9e 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -310,6 +310,16 @@ model LiteLLM_VerificationToken { litellm_budget_table LiteLLM_BudgetTable? @relation(fields: [budget_id], references: [budget_id]) litellm_organization_table LiteLLM_OrganizationTable? @relation(fields: [organization_id], references: [organization_id]) object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) + + // SELECT COUNT(*) FROM (SELECT "public"."LiteLLM_VerificationToken"."token" FROM "public"."LiteLLM_VerificationToken" WHERE ("public"."LiteLLM_VerificationToken"."user_id" = $1 AND ("public"."LiteLLM_VerificationToken"."team_id" IS NULL OR "public"."LiteLLM_VerificationToken"."team_id" <> $2)) OFFSET $3 ) AS "sub" + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."user_id" = $1 OFFSET $2 + @@index([user_id, team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."team_id" = $1 OFFSET $2 + @@index([team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE (("public"."LiteLLM_VerificationToken"."expires" IS NULL OR "public"."LiteLLM_VerificationToken"."expires" > $1) AND "public"."LiteLLM_VerificationToken"."budget_reset_at" < $2) OFFSET $3 + @@index([budget_reset_at, expires]) } // Audit table for deleted keys - preserves spend and key information for historical tracking diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 279946f78de..1750efed92c 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -308,6 +308,16 @@ model LiteLLM_VerificationToken { litellm_budget_table LiteLLM_BudgetTable? @relation(fields: [budget_id], references: [budget_id]) litellm_organization_table LiteLLM_OrganizationTable? @relation(fields: [organization_id], references: [organization_id]) object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) + + // SELECT COUNT(*) FROM (SELECT "public"."LiteLLM_VerificationToken"."token" FROM "public"."LiteLLM_VerificationToken" WHERE ("public"."LiteLLM_VerificationToken"."user_id" = $1 AND ("public"."LiteLLM_VerificationToken"."team_id" IS NULL OR "public"."LiteLLM_VerificationToken"."team_id" <> $2)) OFFSET $3 ) AS "sub" + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."user_id" = $1 OFFSET $2 + @@index([user_id, team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."team_id" = $1 OFFSET $2 + @@index([team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE (("public"."LiteLLM_VerificationToken"."expires" IS NULL OR "public"."LiteLLM_VerificationToken"."expires" > $1) AND "public"."LiteLLM_VerificationToken"."budget_reset_at" < $2) OFFSET $3 + @@index([budget_reset_at, expires]) } // Audit table for deleted keys - preserves spend and key information for historical tracking diff --git a/schema.prisma b/schema.prisma index ecf4e06ef6e..9a87a491cf7 100644 --- a/schema.prisma +++ b/schema.prisma @@ -310,6 +310,16 @@ model LiteLLM_VerificationToken { litellm_budget_table LiteLLM_BudgetTable? @relation(fields: [budget_id], references: [budget_id]) litellm_organization_table LiteLLM_OrganizationTable? @relation(fields: [organization_id], references: [organization_id]) object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) + + // SELECT COUNT(*) FROM (SELECT "public"."LiteLLM_VerificationToken"."token" FROM "public"."LiteLLM_VerificationToken" WHERE ("public"."LiteLLM_VerificationToken"."user_id" = $1 AND ("public"."LiteLLM_VerificationToken"."team_id" IS NULL OR "public"."LiteLLM_VerificationToken"."team_id" <> $2)) OFFSET $3 ) AS "sub" + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."user_id" = $1 OFFSET $2 + @@index([user_id, team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE "public"."LiteLLM_VerificationToken"."team_id" = $1 OFFSET $2 + @@index([team_id]) + + // SELECT ... FROM "public"."LiteLLM_VerificationToken" WHERE (("public"."LiteLLM_VerificationToken"."expires" IS NULL OR "public"."LiteLLM_VerificationToken"."expires" > $1) AND "public"."LiteLLM_VerificationToken"."budget_reset_at" < $2) OFFSET $3 + @@index([budget_reset_at, expires]) } // Audit table for deleted keys - preserves spend and key information for historical tracking From 3d49388d8e2f5ef1fe42bf4548ffcb687c8e826f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 13:36:01 +0530 Subject: [PATCH 35/50] Fix structured response of tool call --- litellm/integrations/websearch_interception/transformation.py | 4 ++-- litellm/llms/openai/openai.py | 1 - 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/websearch_interception/transformation.py b/litellm/integrations/websearch_interception/transformation.py index 0884d408c84..3201d99d69a 100644 --- a/litellm/integrations/websearch_interception/transformation.py +++ b/litellm/integrations/websearch_interception/transformation.py @@ -3,7 +3,7 @@ WebSearch Tool Transformation Transforms between Anthropic/OpenAI tool_use format and LiteLLM search format. """ - +import json from typing import Any, Dict, List, Tuple from litellm._logging import verbose_logger @@ -301,7 +301,7 @@ class WebSearchTransformation: "type": "function", "function": { "name": tc["name"], - "arguments": str(tc["input"]), + "arguments": json.dumps(tc["input"]) if isinstance(tc["input"], dict) else str(tc["input"]), }, } for tc in tool_calls diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index c6f502d3a25..da87852dff5 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -926,7 +926,6 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): logging_obj=logging_obj, ) stringified_response = response.model_dump() - print(f"πŸ”₯stringified_response: {stringified_response}") logging_obj.post_call( input=data["messages"], api_key=api_key, From 4e94ecb08d7fdeaaaff14d239b7bc38d23f53466 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 13:41:29 +0530 Subject: [PATCH 36/50] Add tests for WebSearch interception with chat completions API --- test_websearch_chat_completion.py | 136 ------ .../test_websearch_chat_completion.py | 398 ++++++++++++++++++ 2 files changed, 398 insertions(+), 136 deletions(-) delete mode 100644 test_websearch_chat_completion.py create mode 100644 tests/test_litellm/integrations/websearch_interception/test_websearch_chat_completion.py diff --git a/test_websearch_chat_completion.py b/test_websearch_chat_completion.py deleted file mode 100644 index e572e4d860c..00000000000 --- a/test_websearch_chat_completion.py +++ /dev/null @@ -1,136 +0,0 @@ -""" -Test script for WebSearch interception with chat completions API. - -This script demonstrates how to use the websearch_interception callback -with litellm.acompletion() for transparent server-side web search execution. -""" -import asyncio -import litellm - -# Enable verbose logging to see what's happening -litellm.set_verbose = True - - -async def test_websearch_chat_completion(): - """Test websearch interception with chat completions API.""" - - # Configure WebSearch interception - litellm.callbacks = ["websearch_interception"] - - print("\n" + "="*80) - print("Testing WebSearch Interception with Chat Completions API") - print("="*80 + "\n") - - # User makes a simple completion call with tools - print("Making request to GPT-4o with litellm_web_search tool...") - print("Question: What's the weather in San Francisco today?") - print("\nExpected behavior:") - print("1. Model calls litellm_web_search tool") - print("2. Server executes web search automatically") - print("3. Server makes follow-up request with search results") - print("4. User gets final answer\n") - - response = await litellm.acompletion( - model="gpt-4o", - messages=[ - {"role": "user", "content": "What's the weather in San Francisco today?"} - ], - tools=[ - { - "type": "function", - "function": { - "name": "litellm_web_search", - "description": "Search the web for information", - "parameters": { - "type": "object", - "properties": { - "query": {"type": "string", "description": "Search query"} - }, - "required": ["query"] - } - } - } - ] - ) - - print("\n" + "-"*80) - print("FINAL RESPONSE:") - print("-"*80) - print(f"\nContent: {response.choices[0].message.content}") - print(f"\nFinish reason: {response.choices[0].finish_reason}") - - # Check if we got tool_calls (should NOT if agentic loop worked) - if hasattr(response.choices[0].message, 'tool_calls') and response.choices[0].message.tool_calls: - print("\n⚠️ WARNING: Got tool_calls in response!") - print("This means the agentic loop did NOT execute automatically.") - print(f"Tool calls: {response.choices[0].message.tool_calls}") - else: - print("\nβœ… SUCCESS: No tool_calls in response!") - print("The agentic loop executed automatically and returned the final answer.") - - print("\n" + "="*80 + "\n") - - -async def test_streaming_websearch(): - """Test websearch interception with streaming.""" - - # Configure WebSearch interception - litellm.callbacks = ["websearch_interception"] - - print("\n" + "="*80) - print("Testing WebSearch Interception with STREAMING") - print("="*80 + "\n") - - print("Making STREAMING request to GPT-4o with litellm_web_search tool...") - print("Question: What are the latest AI news?") - - response = await litellm.acompletion( - model="gpt-4o", - messages=[ - {"role": "user", "content": "What are the latest AI news from today?"} - ], - tools=[ - { - "type": "function", - "function": { - "name": "litellm_web_search", - "description": "Search the web for information", - "parameters": { - "type": "object", - "properties": { - "query": {"type": "string"} - } - } - } - } - ], - stream=True - ) - - print("\n" + "-"*80) - print("STREAMING RESPONSE:") - print("-"*80 + "\n") - - full_content = "" - async for chunk in response: - if hasattr(chunk.choices[0].delta, 'content') and chunk.choices[0].delta.content: - content = chunk.choices[0].delta.content - print(content, end="", flush=True) - full_content += content - - print("\n\nβœ… Streaming completed successfully!") - print(f"Total content length: {len(full_content)} chars") - print("\n" + "="*80 + "\n") - - -if __name__ == "__main__": - print("\nWebSearch Interception Test Suite") - print("==================================\n") - print("This test demonstrates transparent server-side web search execution.") - print("The agentic loop happens automatically - user just gets the final answer.\n") - - # Run tests - asyncio.run(test_websearch_chat_completion()) - - # Uncomment to test streaming - # asyncio.run(test_streaming_websearch()) diff --git a/tests/test_litellm/integrations/websearch_interception/test_websearch_chat_completion.py b/tests/test_litellm/integrations/websearch_interception/test_websearch_chat_completion.py new file mode 100644 index 00000000000..1b53633484d --- /dev/null +++ b/tests/test_litellm/integrations/websearch_interception/test_websearch_chat_completion.py @@ -0,0 +1,398 @@ +""" +Integration tests for WebSearch interception with chat completions API. + +Tests the end-to-end flow of websearch_interception callback with +litellm.acompletion() for transparent server-side web search execution. +""" +import os +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +import litellm +from litellm.integrations.websearch_interception.handler import ( + WebSearchInterceptionLogger, +) +from litellm.types.utils import LlmProviders, ModelResponse + + +@pytest.fixture +def mock_search_response(): + """Mock search response from litellm.asearch()""" + mock_response = MagicMock() + mock_response.results = [ + MagicMock( + title="Weather in San Francisco", + url="https://weather.com/sf", + snippet="Current weather: 65Β°F, partly cloudy", + ) + ] + return mock_response + + +@pytest.fixture +def websearch_logger(): + """Create a WebSearchInterceptionLogger instance""" + return WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.OPENAI, LlmProviders.MINIMAX] + ) + + +@pytest.mark.asyncio +@pytest.mark.skipif( + os.environ.get("OPENAI_API_KEY") is None, + reason="OPENAI_API_KEY not set", +) +async def test_websearch_chat_completion_with_openai(): + """Test websearch interception with OpenAI chat completions API. + + This test verifies that: + 1. Model calls litellm_web_search tool + 2. Server executes web search automatically + 3. Server makes follow-up request with search results + 4. User gets final answer without tool_calls + """ + # Configure WebSearch interception + original_callbacks = litellm.callbacks.copy() if litellm.callbacks else [] + websearch_logger = WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.OPENAI] + ) + litellm.callbacks = [websearch_logger] + + try: + response = await litellm.acompletion( + model="gpt-4o-mini", # Use cheaper model for testing + messages=[ + {"role": "user", "content": "What's the weather in San Francisco today?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web for information", + "parameters": { + "type": "object", + "properties": { + "query": { + "type": "string", + "description": "Search query", + } + }, + "required": ["query"], + }, + }, + } + ], + ) + + # Verify response structure + assert isinstance(response, ModelResponse) + assert response.choices[0].message.content is not None + assert len(response.choices[0].message.content) > 0 + + # If agentic loop worked, we should NOT have tool_calls in final response + # (they should have been executed and replaced with final answer) + if hasattr(response.choices[0].message, "tool_calls"): + # If tool_calls exist, it means agentic loop didn't run + # This could happen if search tool is not configured + pytest.skip( + "Agentic loop did not execute - search tool may not be configured" + ) + + # Verify we got a meaningful response + assert response.choices[0].finish_reason in ["stop", "end_turn"] + + finally: + # Restore original callbacks + litellm.callbacks = original_callbacks + + +@pytest.mark.asyncio +async def test_websearch_chat_completion_hook_detection(): + """Test that websearch hook correctly detects tool calls in response.""" + from litellm.types.utils import ( + ChatCompletionMessageToolCall, + Choices, + Function, + Message, + ) + + websearch_logger = WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.OPENAI] + ) + + # Mock response with litellm_web_search tool call + mock_response = ModelResponse( + id="test-123", + choices=[ + Choices( + finish_reason="tool_calls", + index=0, + message=Message( + role="assistant", + content=None, + tool_calls=[ + ChatCompletionMessageToolCall( + id="call_123", + type="function", + function=Function( + name="litellm_web_search", + arguments='{"query": "weather in SF"}', + ), + ) + ], + ) + ) + ], + model="gpt-4o", + object="chat.completion", + created=1234567890, + ) + + # Test should_run_chat_completion_agentic_loop + should_run, tools_dict = ( + await websearch_logger.async_should_run_chat_completion_agentic_loop( + response=mock_response, + model="gpt-4o", + messages=[{"role": "user", "content": "What's the weather?"}], + tools=[ + { + "type": "function", + "function": {"name": "litellm_web_search"}, + } + ], + stream=False, + custom_llm_provider="openai", + kwargs={}, + ) + ) + + # Verify hook detected the tool call + assert should_run is True + assert "tool_calls" in tools_dict + assert len(tools_dict["tool_calls"]) == 1 + assert tools_dict["tool_calls"][0]["name"] == "litellm_web_search" + assert tools_dict["response_format"] == "openai" + + +@pytest.mark.asyncio +async def test_websearch_not_triggered_without_tool(): + """Test that websearch hook is NOT triggered when no web search tool in request.""" + from litellm.types.utils import Choices, Message + + websearch_logger = WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.OPENAI] + ) + + mock_response = ModelResponse( + id="test-123", + choices=[ + Choices( + finish_reason="stop", + index=0, + message=Message( + role="assistant", + content="Here's the answer", + tool_calls=None, + ) + ) + ], + model="gpt-4o", + object="chat.completion", + created=1234567890, + ) + + # Test without web search tool + should_run, tools_dict = ( + await websearch_logger.async_should_run_chat_completion_agentic_loop( + response=mock_response, + model="gpt-4o", + messages=[{"role": "user", "content": "Hello"}], + tools=[ + { + "type": "function", + "function": {"name": "some_other_tool"}, + } + ], + stream=False, + custom_llm_provider="openai", + kwargs={}, + ) + ) + + # Verify hook did NOT trigger + assert should_run is False + assert tools_dict == {} + + +@pytest.mark.asyncio +async def test_websearch_not_triggered_for_disabled_provider(): + """Test that websearch hook is NOT triggered for providers not in enabled_providers.""" + from litellm.types.utils import ( + ChatCompletionMessageToolCall, + Choices, + Function, + Message, + ) + + # Only enable bedrock + websearch_logger = WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.BEDROCK] + ) + + mock_response = ModelResponse( + id="test-123", + choices=[ + Choices( + finish_reason="tool_calls", + index=0, + message=Message( + role="assistant", + content=None, + tool_calls=[ + ChatCompletionMessageToolCall( + id="call_123", + type="function", + function=Function( + name="litellm_web_search", + arguments='{"query": "test"}', + ), + ) + ], + ) + ) + ], + model="gpt-4o", + object="chat.completion", + created=1234567890, + ) + + # Test with OpenAI provider (not enabled) + should_run, tools_dict = ( + await websearch_logger.async_should_run_chat_completion_agentic_loop( + response=mock_response, + model="gpt-4o", + messages=[{"role": "user", "content": "test"}], + tools=[ + { + "type": "function", + "function": {"name": "litellm_web_search"}, + } + ], + stream=False, + custom_llm_provider="openai", # Not in enabled_providers + kwargs={}, + ) + ) + + # Verify hook did NOT trigger + assert should_run is False + assert tools_dict == {} + + +@pytest.mark.asyncio +async def test_websearch_json_serialization_fix(): + """Test that tool call arguments are properly JSON serialized. + + Regression test for the bug where arguments were converted to Python + string representation instead of proper JSON, causing providers like + MiniMax to reject requests with 'invalid function arguments json string'. + """ + from litellm.integrations.websearch_interception.transformation import ( + WebSearchTransformation, + ) + + # Mock tool calls with dict input + tool_calls = [ + { + "id": "call_123", + "name": "litellm_web_search", + "input": {"query": "weather in SF"}, # Dict input + } + ] + + search_results = ["Weather: 65Β°F, partly cloudy"] + + # Transform to OpenAI format + assistant_message, tool_messages = WebSearchTransformation.transform_response( + tool_calls=tool_calls, + search_results=search_results, + response_format="openai", + ) + + # Verify arguments are properly JSON serialized + import json + + arguments_str = assistant_message["tool_calls"][0]["function"]["arguments"] + + # Should be valid JSON + parsed_args = json.loads(arguments_str) + assert parsed_args == {"query": "weather in SF"} + + # Should NOT be Python string representation like "{'query': 'weather in SF'}" + assert arguments_str == '{"query": "weather in SF"}' + assert arguments_str != "{'query': 'weather in SF'}" + + +@pytest.mark.asyncio +@pytest.mark.skipif( + os.environ.get("OPENAI_API_KEY") is None + or os.environ.get("PERPLEXITY_API_KEY") is None, + reason="OPENAI_API_KEY or PERPLEXITY_API_KEY not set", +) +async def test_websearch_streaming_conversion(): + """Test that streaming requests are converted to non-streaming for web search. + + When stream=True is passed with web search tools, the handler should: + 1. Convert stream=True to stream=False for initial request + 2. Execute web search + 3. Convert final response back to streaming + """ + websearch_logger = WebSearchInterceptionLogger( + enabled_providers=[LlmProviders.OPENAI], search_tool_name="perplexity-search" + ) + litellm.callbacks = [websearch_logger] + + try: + response = await litellm.acompletion( + model="gpt-4o-mini", + messages=[ + {"role": "user", "content": "What's the latest AI news?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web", + "parameters": { + "type": "object", + "properties": {"query": {"type": "string"}}, + }, + }, + } + ], + stream=True, + ) + + # Response should be a streaming iterator + chunks = [] + async for chunk in response: + chunks.append(chunk) + + # Verify we got streaming chunks + assert len(chunks) > 0 + + # Verify chunks have expected structure + for chunk in chunks: + assert hasattr(chunk, "choices") + assert len(chunk.choices) > 0 + + finally: + litellm.callbacks = [] + + +if __name__ == "__main__": + # Run with: pytest test_websearch_chat_completion.py -v -s + pytest.main([__file__, "-v", "-s"]) From 7fa4d090ece9278eb4247bbbb896cabacb93eeac Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 13:51:26 +0530 Subject: [PATCH 37/50] Add doc for chat completion web search --- .../integrations/websearch_interception.md | 411 ++++++++++++++++++ docs/my-website/sidebars.js | 5 + 2 files changed, 416 insertions(+) create mode 100644 docs/my-website/docs/integrations/websearch_interception.md diff --git a/docs/my-website/docs/integrations/websearch_interception.md b/docs/my-website/docs/integrations/websearch_interception.md new file mode 100644 index 00000000000..0c5d8927013 --- /dev/null +++ b/docs/my-website/docs/integrations/websearch_interception.md @@ -0,0 +1,411 @@ +# Web Search Integration + +Enable transparent server-side web search execution for any LLM provider. LiteLLM automatically intercepts web search tool calls and executes them using your configured search provider (Perplexity, Tavily, etc.). + +## Quick Start + +### 1. Configure Web Search Interception + +Add to your `config.yaml`: + +```yaml +model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + callbacks: + - websearch_interception: + enabled_providers: + - openai + - minimax + - anthropic + search_tool_name: perplexity-search # Optional + +search_tools: + - search_tool_name: perplexity-search + litellm_params: + search_provider: perplexity + api_key: os.environ/PERPLEXITY_API_KEY +``` + +### 2. Use with Any Provider + +```python +import litellm + +response = await litellm.acompletion( + model="gpt-4o", + messages=[ + {"role": "user", "content": "What's the weather in San Francisco today?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web for information", + "parameters": { + "type": "object", + "properties": { + "query": {"type": "string", "description": "Search query"} + }, + "required": ["query"] + } + } + } + ] +) + +# Response includes search results automatically! +print(response.choices[0].message.content) +``` + +## How It Works + +When a model makes a web search tool call, LiteLLM: + +1. **Detects** the `litellm_web_search` tool call in the response +2. **Executes** the search using your configured search provider +3. **Makes a follow-up request** with the search results +4. **Returns** the final answer to the user + +```mermaid +sequenceDiagram + participant User + participant LiteLLM + participant LLM as LLM Provider + participant Search as Search Provider + + User->>LiteLLM: Request with web_search tool + LiteLLM->>LLM: Forward request + LLM-->>LiteLLM: Response with tool_call + Note over LiteLLM: Detect web search
tool call + LiteLLM->>Search: Execute search + Search-->>LiteLLM: Search results + LiteLLM->>LLM: Follow-up with results + LLM-->>LiteLLM: Final answer + LiteLLM-->>User: Final answer with search results +``` + +**Result**: One API call from user β†’ Complete answer with search results + +## Supported Providers + +Web search integration works with **all providers** that use: +- βœ… **Base HTTP Handler** (`BaseLLMHTTPHandler`) +- βœ… **OpenAI Completion Handler** (`OpenAIChatCompletion`) + +### Providers Using Base HTTP Handler + +| Provider | Status | Notes | +|----------|--------|-------| +| **OpenAI** | βœ… Supported | GPT-4, GPT-3.5, etc. | +| **Anthropic** | βœ… Supported | Claude models via HTTP handler | +| **MiniMax** | βœ… Supported | All MiniMax models | +| **Mistral** | βœ… Supported | Mistral AI models | +| **Cohere** | βœ… Supported | Command models | +| **Fireworks AI** | βœ… Supported | All Fireworks models | +| **Together AI** | βœ… Supported | All Together AI models | +| **Groq** | βœ… Supported | All Groq models | +| **Perplexity** | βœ… Supported | Perplexity models | +| **DeepSeek** | βœ… Supported | DeepSeek models | +| **xAI** | βœ… Supported | Grok models | +| **Hugging Face** | βœ… Supported | Inference API models | +| **OCI** | βœ… Supported | Oracle Cloud models | +| **Vertex AI** | βœ… Supported | Google Vertex AI models | +| **Bedrock** | βœ… Supported | AWS Bedrock models (converse_like route) | +| **Azure OpenAI** | βœ… Supported | Azure-hosted OpenAI models | +| **Sagemaker** | βœ… Supported | AWS Sagemaker models | +| **Databricks** | βœ… Supported | Databricks models | +| **DataRobot** | βœ… Supported | DataRobot models | +| **Hosted VLLM** | βœ… Supported | Self-hosted VLLM | +| **Heroku** | βœ… Supported | Heroku-hosted models | +| **RAGFlow** | βœ… Supported | RAGFlow models | +| **Compactif** | βœ… Supported | Compactif models | +| **Cometapi** | βœ… Supported | Comet API models | +| **A2A** | βœ… Supported | Agent-to-Agent models | +| **Bytez** | βœ… Supported | Bytez models | + +### Providers Using OpenAI Handler + +| Provider | Status | Notes | +|----------|--------|-------| +| **OpenAI** | βœ… Supported | Native OpenAI API | +| **Azure OpenAI** | βœ… Supported | Azure-hosted OpenAI | +| **OpenAI-Compatible** | βœ… Supported | Any OpenAI-compatible API | + +## Configuration + +### WebSearch Interception Parameters + +| Parameter | Type | Required | Description | Example | +|-----------|------|----------|-------------|---------| +| `enabled_providers` | List[String] | Yes | List of providers to enable web search for | `[openai, minimax, anthropic]` | +| `search_tool_name` | String | No | Specific search tool from `search_tools` config. If not set, uses first available. | `perplexity-search` | + +### Provider Values + +Use these values in `enabled_providers`: + +| Provider | Value | Provider | Value | +|----------|-------|----------|-------| +| OpenAI | `openai` | Anthropic | `anthropic` | +| MiniMax | `minimax` | Mistral | `mistral` | +| Cohere | `cohere` | Fireworks AI | `fireworks_ai` | +| Together AI | `together_ai` | Groq | `groq` | +| Perplexity | `perplexity` | DeepSeek | `deepseek` | +| xAI | `xai` | Hugging Face | `huggingface` | +| OCI | `oci` | Vertex AI | `vertex_ai` | +| Bedrock | `bedrock` | Azure | `azure` | +| Sagemaker | `sagemaker_chat` | Databricks | `databricks` | +| DataRobot | `datarobot` | VLLM | `hosted_vllm` | +| Heroku | `heroku` | RAGFlow | `ragflow` | +| Compactif | `compactif` | Cometapi | `cometapi` | +| A2A | `a2a` | Bytez | `bytez` | + +## Search Providers + +Configure which search provider to use. LiteLLM supports multiple search providers: + +| Provider | `search_provider` Value | Environment Variable | +|----------|------------------------|----------------------| +| **Perplexity AI** | `perplexity` | `PERPLEXITYAI_API_KEY` | +| **Tavily** | `tavily` | `TAVILY_API_KEY` | +| **Exa AI** | `exa_ai` | `EXA_API_KEY` | +| **Parallel AI** | `parallel_ai` | `PARALLEL_AI_API_KEY` | +| **Google PSE** | `google_pse` | `GOOGLE_PSE_API_KEY`, `GOOGLE_PSE_ENGINE_ID` | +| **DataForSEO** | `dataforseo` | `DATAFORSEO_LOGIN`, `DATAFORSEO_PASSWORD` | +| **Firecrawl** | `firecrawl` | `FIRECRAWL_API_KEY` | +| **SearXNG** | `searxng` | `SEARXNG_API_BASE` (required) | +| **Linkup** | `linkup` | `LINKUP_API_KEY` | + +See [Search Providers Documentation](../search/index.md) for detailed setup instructions. + +## Complete Configuration Example + +```yaml +model_list: + # OpenAI + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + + # MiniMax + - model_name: minimax + litellm_params: + model: minimax/MiniMax-M2.1 + api_key: os.environ/MINIMAX_API_KEY + + # Anthropic + - model_name: claude + litellm_params: + model: anthropic/claude-sonnet-4-5 + api_key: os.environ/ANTHROPIC_API_KEY + + # Azure OpenAI + - model_name: azure-gpt4 + litellm_params: + model: azure/gpt-4 + api_base: https://my-azure.openai.azure.com + api_key: os.environ/AZURE_API_KEY + +litellm_settings: + callbacks: + - websearch_interception: + enabled_providers: + - openai + - minimax + - anthropic + - azure + search_tool_name: perplexity-search + +search_tools: + - search_tool_name: perplexity-search + litellm_params: + search_provider: perplexity + api_key: os.environ/PERPLEXITY_API_KEY + + - search_tool_name: tavily-search + litellm_params: + search_provider: tavily + api_key: os.environ/TAVILY_API_KEY +``` + +## Usage Examples + +### Python SDK + +```python +import litellm + +# Configure callbacks +litellm.callbacks = ["websearch_interception"] + +# Make completion with web search tool +response = await litellm.acompletion( + model="gpt-4o", + messages=[ + {"role": "user", "content": "What are the latest AI news?"} + ], + tools=[ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web for current information", + "parameters": { + "type": "object", + "properties": { + "query": { + "type": "string", + "description": "Search query" + } + }, + "required": ["query"] + } + } + } + ] +) + +print(response.choices[0].message.content) +``` + +### Proxy Server + +```bash +# Start proxy with config +litellm --config config.yaml + +# Make request +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "model": "gpt-4o", + "messages": [ + {"role": "user", "content": "What is the weather in San Francisco?"} + ], + "tools": [ + { + "type": "function", + "function": { + "name": "litellm_web_search", + "description": "Search the web", + "parameters": { + "type": "object", + "properties": { + "query": {"type": "string"} + }, + "required": ["query"] + } + } + } + ] + }' +``` + +## How Search Tool Selection Works + +1. **If `search_tool_name` is specified** β†’ Uses that specific search tool +2. **If `search_tool_name` is not specified** β†’ Uses first search tool in `search_tools` list + +```yaml +search_tools: + - search_tool_name: perplexity-search # ← This will be used if no search_tool_name specified + litellm_params: + search_provider: perplexity + api_key: os.environ/PERPLEXITY_API_KEY + + - search_tool_name: tavily-search + litellm_params: + search_provider: tavily + api_key: os.environ/TAVILY_API_KEY +``` + +## Troubleshooting + +### Web Search Not Working + +1. **Check provider is enabled**: + ```yaml + enabled_providers: + - openai # Make sure your provider is in this list + ``` + +2. **Verify search tool is configured**: + ```yaml + search_tools: + - search_tool_name: perplexity-search + litellm_params: + search_provider: perplexity + api_key: os.environ/PERPLEXITY_API_KEY + ``` + +3. **Check API keys are set**: + ```bash + export PERPLEXITY_API_KEY=your-key + ``` + +4. **Enable debug logging**: + ```python + litellm.set_verbose = True + ``` + +### Common Issues + +**Issue**: Model returns tool_calls instead of final answer +- **Cause**: Provider not in `enabled_providers` list +- **Solution**: Add provider to `enabled_providers` + +**Issue**: "No search tool configured" error +- **Cause**: No search tools in `search_tools` config +- **Solution**: Add at least one search tool configuration + +**Issue**: "Invalid function arguments json string" error (MiniMax) +- **Cause**: Fixed in latest version - arguments weren't properly JSON serialized +- **Solution**: Update to latest LiteLLM version + +## Related Documentation + +- [Search Providers](../search/index.md) - Detailed search provider setup +- [Claude Code WebSearch](../tutorials/claude_code_websearch.md) - Using with Claude Code +- [Tool Calling](../completion/function_call.md) - General tool calling documentation +- [Callbacks](./custom_callback.md) - Custom callback documentation + +## Technical Details + +### Architecture + +Web search integration is implemented as a custom callback (`WebSearchInterceptionLogger`) that: + +1. **Pre-request Hook**: Converts native web search tools to LiteLLM standard format +2. **Post-response Hook**: Detects web search tool calls in responses +3. **Agentic Loop**: Executes searches and makes follow-up requests automatically + +### Supported APIs + +- βœ… **Chat Completions API** (OpenAI format) +- βœ… **Anthropic Messages API** (Anthropic format) +- βœ… **Streaming** (automatically converted) +- βœ… **Non-streaming** + +### Response Format Detection + +The handler automatically detects response format: +- **OpenAI format**: `tool_calls` in assistant message +- **Anthropic format**: `tool_use` blocks in content + +### Performance + +- **Latency**: Adds one additional LLM call (follow-up request with search results) +- **Caching**: Search results can be cached (depends on search provider) +- **Parallel Searches**: Multiple search queries executed in parallel + +## Contributing + +Found a bug or want to add support for a new provider? See our [Contributing Guide](https://github.com/BerriAI/litellm/blob/main/CONTRIBUTING.md). diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index fda0e3be4e4..9bedb228393 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -96,6 +96,11 @@ const sidebars = { "proxy/prometheus" ] }, + { + type: "doc", + id: "integrations/websearch_interception", + label: "Web Search Integration" + }, { type: "category", label: "[Beta] Prompt Management", From 1f04115fb0ff939f42952d631504ac9aa96b5811 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 13:59:05 +0530 Subject: [PATCH 38/50] Fix: is_web_search_tool_chat_completion --- .../websearch_interception/handler.py | 7 +-- .../websearch_interception/tools.py | 44 +++++++++++++++++++ 2 files changed, 48 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 1e109dc9e39..7e8fa66c493 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -17,6 +17,7 @@ from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.websearch_interception.tools import ( get_litellm_web_search_tool, is_web_search_tool, + is_web_search_tool_chat_completion, ) from litellm.integrations.websearch_interception.transformation import ( WebSearchTransformation, @@ -325,11 +326,11 @@ class WebSearchInterceptionLogger(CustomLogger): ) return False, {} - # Check if tools include any web search tool - has_websearch_tool = any(is_web_search_tool(t) for t in (tools or [])) + # Check if tools include any web search tool (strict check for chat completions) + has_websearch_tool = any(is_web_search_tool_chat_completion(t) for t in (tools or [])) if not has_websearch_tool: verbose_logger.debug( - "WebSearchInterception: No web search tool in request" + "WebSearchInterception: No litellm_web_search tool in request" ) return False, {} diff --git a/litellm/integrations/websearch_interception/tools.py b/litellm/integrations/websearch_interception/tools.py index be8808622da..c39d150fb19 100644 --- a/litellm/integrations/websearch_interception/tools.py +++ b/litellm/integrations/websearch_interception/tools.py @@ -49,6 +49,50 @@ def get_litellm_web_search_tool() -> Dict[str, Any]: } +def is_web_search_tool_chat_completion(tool: Dict[str, Any]) -> bool: + """ + Check if a tool is a web search tool for Chat Completions API (strict check). + + This is a stricter version that ONLY checks for the exact LiteLLM web search tool name. + Use this for Chat Completions API to avoid false positives with user-defined tools. + + Detects ONLY: + - LiteLLM standard: name == "litellm_web_search" (Anthropic format) + - OpenAI format: type == "function" with function.name == "litellm_web_search" + + Args: + tool: Tool dictionary to check + + Returns: + True if tool is exactly the LiteLLM web search tool + + Example: + >>> is_web_search_tool_chat_completion({"name": "litellm_web_search"}) + True + >>> is_web_search_tool_chat_completion({"type": "function", "function": {"name": "litellm_web_search"}}) + True + >>> is_web_search_tool_chat_completion({"name": "web_search"}) + False + >>> is_web_search_tool_chat_completion({"name": "WebSearch"}) + False + """ + tool_name = tool.get("name", "") + tool_type = tool.get("type", "") + + # Check for OpenAI format: {"type": "function", "function": {"name": "litellm_web_search"}} + if tool_type == "function" and "function" in tool: + function_def = tool.get("function", {}) + function_name = function_def.get("name", "") + if function_name == LITELLM_WEB_SEARCH_TOOL_NAME: + return True + + # Check for LiteLLM standard tool (Anthropic format) + if tool_name == LITELLM_WEB_SEARCH_TOOL_NAME: + return True + + return False + + def is_web_search_tool(tool: Dict[str, Any]) -> bool: """ Check if a tool is a web search tool (native or LiteLLM standard). From c48986ba8d2d845b782e8ed923a6405263d2b9ee Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 14:21:43 +0530 Subject: [PATCH 39/50] Fix double json import --- litellm/integrations/websearch_interception/transformation.py | 1 - 1 file changed, 1 deletion(-) diff --git a/litellm/integrations/websearch_interception/transformation.py b/litellm/integrations/websearch_interception/transformation.py index 3201d99d69a..ad92f5d03d0 100644 --- a/litellm/integrations/websearch_interception/transformation.py +++ b/litellm/integrations/websearch_interception/transformation.py @@ -190,7 +190,6 @@ class WebSearchTransformation: LITELLM_WEB_SEARCH_TOOL_NAME, "WebSearch", "web_search" ): # Parse arguments (might be JSON string) - import json if isinstance(function_arguments, str): try: arguments = json.loads(function_arguments) From 6c4d6bb15e67a8d05a78376e020a6622ef85dc5c Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 16:00:01 +0530 Subject: [PATCH 40/50] Add new vercel ai anthropic models --- ...odel_prices_and_context_window_backup.json | 187 ++++++++++++++++++ model_prices_and_context_window.json | 187 ++++++++++++++++++ 2 files changed, 374 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5cad0db241f..259fb656457 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -28567,6 +28567,193 @@ "supports_function_calling": true, "supports_tool_choice": true }, + "vercel_ai_gateway/anthropic/claude-3-5-sonnet": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-3-5-sonnet-20241022": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-3-7-sonnet": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-haiku-4.5": { + "cache_creation_input_token_cost": 1.25e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 5e-06, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4": { + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "input_cost_per_token": 1.5e-05, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "max_tokens": 32000, + "mode": "chat", + "output_cost_per_token": 7.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.1": { + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "input_cost_per_token": 1.5e-05, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "max_tokens": 32000, + "mode": "chat", + "output_cost_per_token": 7.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.5": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 2.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.6": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 2.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-sonnet-4": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-sonnet-4.5": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 1000000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, "vercel_ai_gateway/cohere/command-a": { "input_cost_per_token": 2.5e-06, "litellm_provider": "vercel_ai_gateway", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5cad0db241f..259fb656457 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -28567,6 +28567,193 @@ "supports_function_calling": true, "supports_tool_choice": true }, + "vercel_ai_gateway/anthropic/claude-3-5-sonnet": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-3-5-sonnet-20241022": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-3-7-sonnet": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-haiku-4.5": { + "cache_creation_input_token_cost": 1.25e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 5e-06, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4": { + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "input_cost_per_token": 1.5e-05, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "max_tokens": 32000, + "mode": "chat", + "output_cost_per_token": 7.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.1": { + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "input_cost_per_token": 1.5e-05, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "max_tokens": 32000, + "mode": "chat", + "output_cost_per_token": 7.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.5": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 2.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-opus-4.6": { + "cache_creation_input_token_cost": 6.25e-06, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 2.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-sonnet-4": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "vercel_ai_gateway/anthropic/claude-sonnet-4.5": { + "cache_creation_input_token_cost": 3.75e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 1000000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 1.5e-05, + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, "vercel_ai_gateway/cohere/command-a": { "input_cost_per_token": 2.5e-06, "litellm_provider": "vercel_ai_gateway", From d35691aa0ce3e40bb4555fed715d9e45ec64e361 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 16:25:46 +0530 Subject: [PATCH 41/50] Fix: base_model name for body and deplyment name in URL --- litellm/llms/azure/azure.py | 18 +- .../test_azure_image_generation_init.py | 205 ++++++++++++++++++ 2 files changed, 214 insertions(+), 9 deletions(-) diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index cb9fe0aeb30..95afa53b625 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -1060,6 +1060,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): headers: dict, client=None, timeout=None, + model: Optional[str] = None, ) -> ImageResponse: response: Optional[dict] = None @@ -1071,8 +1072,9 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): if api_base.endswith("/"): api_base = api_base.rstrip("/") api_version: str = azure_client_params.get("api_version", "") + # Use the deployment name (model) for URL construction, not the base_model from data img_gen_api_base = self.create_azure_base_url( - azure_client_params=azure_client_params, model=data.get("model", "") + azure_client_params=azure_client_params, model=model or data.get("model", "") ) ## LOGGING @@ -1159,21 +1161,18 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): model = model else: model = None - ## BASE MODEL CHECK if ( model_response is not None - and optional_params.get("base_model", None) is not None + and litellm_params.get("base_model", None) is not None ): - model_response._hidden_params["model"] = optional_params.pop( - "base_model" - ) + model_response._hidden_params["model"] = litellm_params.get("base_model", None) # Azure image generation API doesn't support extra_body parameter extra_body = optional_params.pop("extra_body", {}) flattened_params = {**optional_params, **extra_body} - data = {"model": model, "prompt": prompt, **flattened_params} + data = {"model": litellm_params.get("base_model", None) or model, "prompt": prompt, **flattened_params} max_retries = data.pop("max_retries", 2) if not isinstance(max_retries, int): raise AzureOpenAIError( @@ -1196,10 +1195,11 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): is_async=False, ) if aimg_generation is True: - return self.aimage_generation(data=data, input=input, logging_obj=logging_obj, model_response=model_response, api_key=api_key, client=client, azure_client_params=azure_client_params, timeout=timeout, headers=headers) # type: ignore + return self.aimage_generation(data=data, input=input, logging_obj=logging_obj, model_response=model_response, api_key=api_key, client=client, azure_client_params=azure_client_params, timeout=timeout, headers=headers, model=model) # type: ignore + # Use the deployment name (model) for URL construction, not the base_model from data img_gen_api_base = self.create_azure_base_url( - azure_client_params=azure_client_params, model=data.get("model", "") + azure_client_params=azure_client_params, model=model ) ## LOGGING diff --git a/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py b/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py index 987eb5bf998..bf8165c6908 100644 --- a/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py +++ b/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py @@ -251,3 +251,208 @@ def test_azure_image_generation_drop_params_false_raises_error(): # Verify the error message mentions the unsupported parameter assert "response_format" in str(exc_info.value) + + +def test_azure_image_generation_base_model_vs_deployment_name(): + """ + Test that Azure image generation correctly uses base_model in request body + but deployment name in the URL. + + When base_model is specified in litellm_params, the request should: + 1. Use base_model (e.g., "gpt-image-1.5") in the JSON request body + 2. Use the deployment name (e.g., "gpt-image-15") in the URL path + + This is important because Azure expects: + - URL: /openai/deployments/{deployment_name}/images/generations + - Body: {"model": "{base_model}", ...} + + Example config: + model: azure/gpt-image-15 # deployment name + base_model: gpt-image-1.5 # actual model name + """ + from unittest.mock import MagicMock + + # Setup test parameters + azure_chat_completion = AzureChatCompletion() + + prompt = "A beautiful image of a cat" + model = "gpt-image-15" # This is the deployment name + base_model = "gpt-image-1.5" # This is the actual model name + api_base = "https://openai-gpt-image-1-5-test-v-1.openai.azure.com/" + api_version = "2024-07-01-preview" + api_key = "test-api-key" + + litellm_params = { + "base_model": base_model, + "api_base": api_base, + "api_version": api_version, + } + + optional_params = { + "n": 1, + "size": "1024x1024" + } + + # Mock the HTTP request to capture what gets sent + with patch.object( + azure_chat_completion, + "make_sync_azure_httpx_request", + return_value=MagicMock( + json=lambda: { + "created": 1234567890, + "data": [ + { + "url": "https://example.com/image.png", + "revised_prompt": prompt + } + ] + } + ) + ) as mock_request: + # Mock logging object + logging_obj = MagicMock() + logging_obj.pre_call = MagicMock() + logging_obj.post_call = MagicMock() + + # Call the image_generation method + try: + response = azure_chat_completion.image_generation( + prompt=prompt, + timeout=60.0, + optional_params=optional_params, + logging_obj=logging_obj, + headers={}, + model=model, + api_key=api_key, + api_base=api_base, + api_version=api_version, + litellm_params=litellm_params, + ) + except Exception as e: + # If there's an error, we still want to check the mock calls + pass + + # Verify the mock was called + assert mock_request.called, "HTTP request should have been made" + + # Get the call arguments + call_kwargs = mock_request.call_args.kwargs + + # Verify the URL uses the deployment name (not base_model) + api_base_used = call_kwargs.get("api_base", "") + assert model in api_base_used, ( + f"URL should contain deployment name '{model}', " + f"but got: {api_base_used}" + ) + assert base_model not in api_base_used or base_model == model, ( + f"URL should NOT contain base_model '{base_model}' when it differs from deployment name, " + f"but got: {api_base_used}" + ) + + # Verify the request body uses base_model (not deployment name) + request_data = call_kwargs.get("data", {}) + assert request_data.get("model") == base_model, ( + f"Request body 'model' field should be base_model '{base_model}', " + f"but got: {request_data.get('model')}" + ) + + # Verify other fields are correct + assert request_data.get("prompt") == prompt + assert request_data.get("n") == 1 + assert request_data.get("size") == "1024x1024" + + +@pytest.mark.asyncio +async def test_azure_aimage_generation_base_model_vs_deployment_name(): + """ + Test that Azure async image generation correctly uses base_model in request body + but deployment name in the URL. + + This is the async version of test_azure_image_generation_base_model_vs_deployment_name. + """ + from unittest.mock import MagicMock + + # Setup test parameters + azure_chat_completion = AzureChatCompletion() + + prompt = "A beautiful image of a cat" + model = "gpt-image-15" # This is the deployment name + base_model = "gpt-image-1.5" # This is the actual model name + api_base = "https://openai-gpt-image-1-5-test-v-1.openai.azure.com/" + api_version = "2024-07-01-preview" + api_key = "test-api-key" + + data = { + "model": base_model, + "prompt": prompt, + "n": 1, + "size": "1024x1024" + } + + azure_client_params = { + "api_base": api_base, + "api_version": api_version, + } + + # Mock the HTTP request to capture what gets sent + with patch.object( + azure_chat_completion, + "make_async_azure_httpx_request", + new_callable=AsyncMock, + return_value=MagicMock( + json=lambda: { + "created": 1234567890, + "data": [ + { + "url": "https://example.com/image.png", + "revised_prompt": prompt + } + ] + } + ) + ) as mock_request: + # Mock logging object + logging_obj = MagicMock() + logging_obj.pre_call = MagicMock() + logging_obj.post_call = MagicMock() + + # Call the aimage_generation method + try: + response = await azure_chat_completion.aimage_generation( + data=data, + model_response=None, + azure_client_params=azure_client_params, + api_key=api_key, + input=[], + logging_obj=logging_obj, + headers={}, + model=model, # Pass the deployment name + timeout=60.0, + ) + except Exception as e: + # If there's an error, we still want to check the mock calls + pass + + # Verify the mock was called + assert mock_request.called, "HTTP request should have been made" + + # Get the call arguments + call_kwargs = mock_request.call_args.kwargs + + # Verify the URL uses the deployment name (not base_model) + api_base_used = call_kwargs.get("api_base", "") + assert model in api_base_used, ( + f"URL should contain deployment name '{model}', " + f"but got: {api_base_used}" + ) + assert base_model not in api_base_used or base_model == model, ( + f"URL should NOT contain base_model '{base_model}' when it differs from deployment name, " + f"but got: {api_base_used}" + ) + + # Verify the request body uses base_model (not deployment name) + request_data = call_kwargs.get("data", {}) + assert request_data.get("model") == base_model, ( + f"Request body 'model' field should be base_model '{base_model}', " + f"but got: {request_data.get('model')}" + ) From 1b2278951d04aeef568714571fba69de56785f6a Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 16:39:21 +0530 Subject: [PATCH 42/50] Add output_config as supported param --- .../experimental_pass_through/messages/transformation.py | 1 + litellm/types/llms/anthropic.py | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index bb40f9df266..a48d1622155 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -46,6 +46,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "thinking", "context_management", "output_format", + "output_config", # TODO: Add Anthropic `metadata` support # "metadata", ] diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index fedf419efd6..6c47a544739 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -360,6 +360,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False): context_management: Optional[Dict[str, Any]] container: Optional[Dict[str, Any]] # Container config with skills for code execution output_format: Optional[AnthropicOutputSchema] # Structured outputs support + output_config: Optional[AnthropicOutputConfig] # Configuration for Claude's output behavior class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False): From 23088f86bd19c7b868a0e90a522a54438aca59a9 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:07:36 +0530 Subject: [PATCH 43/50] Add response schema for vercel ai sonnet 4.5 --- litellm/model_prices_and_context_window_backup.json | 3 ++- model_prices_and_context_window.json | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 259fb656457..6cc7f737ce4 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -28763,7 +28763,8 @@ "mode": "chat", "output_cost_per_token": 1e-05, "supports_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_response_schema": true }, "vercel_ai_gateway/cohere/command-r": { "input_cost_per_token": 1.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 259fb656457..6cc7f737ce4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -28763,7 +28763,8 @@ "mode": "chat", "output_cost_per_token": 1e-05, "supports_function_calling": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_response_schema": true }, "vercel_ai_gateway/cohere/command-r": { "input_cost_per_token": 1.5e-07, From 30d17c29e4bcff4c5843fd1befb2dfb9d3832fa0 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:13:39 +0530 Subject: [PATCH 44/50] handle when litellm_parrams might be none --- litellm/llms/azure/azure.py | 4 +- .../test_azure_image_generation_init.py | 54 ++++++++----------- 2 files changed, 26 insertions(+), 32 deletions(-) diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index 95afa53b625..76fa713ca8c 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -1164,6 +1164,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): ## BASE MODEL CHECK if ( model_response is not None + and litellm_params is not None and litellm_params.get("base_model", None) is not None ): model_response._hidden_params["model"] = litellm_params.get("base_model", None) @@ -1172,7 +1173,8 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): extra_body = optional_params.pop("extra_body", {}) flattened_params = {**optional_params, **extra_body} - data = {"model": litellm_params.get("base_model", None) or model, "prompt": prompt, **flattened_params} + base_model = litellm_params.get("base_model", None) if litellm_params else None + data = {"model": base_model or model, "prompt": prompt, **flattened_params} max_retries = data.pop("max_retries", 2) if not isinstance(max_retries, int): raise AzureOpenAIError( diff --git a/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py b/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py index bf8165c6908..44bcc9f954a 100644 --- a/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py +++ b/tests/test_litellm/llms/azure/image_generation/test_azure_image_generation_init.py @@ -315,22 +315,18 @@ def test_azure_image_generation_base_model_vs_deployment_name(): logging_obj.post_call = MagicMock() # Call the image_generation method - try: - response = azure_chat_completion.image_generation( - prompt=prompt, - timeout=60.0, - optional_params=optional_params, - logging_obj=logging_obj, - headers={}, - model=model, - api_key=api_key, - api_base=api_base, - api_version=api_version, - litellm_params=litellm_params, - ) - except Exception as e: - # If there's an error, we still want to check the mock calls - pass + response = azure_chat_completion.image_generation( + prompt=prompt, + timeout=60.0, + optional_params=optional_params, + logging_obj=logging_obj, + headers={}, + model=model, + api_key=api_key, + api_base=api_base, + api_version=api_version, + litellm_params=litellm_params, + ) # Verify the mock was called assert mock_request.called, "HTTP request should have been made" @@ -417,21 +413,17 @@ async def test_azure_aimage_generation_base_model_vs_deployment_name(): logging_obj.post_call = MagicMock() # Call the aimage_generation method - try: - response = await azure_chat_completion.aimage_generation( - data=data, - model_response=None, - azure_client_params=azure_client_params, - api_key=api_key, - input=[], - logging_obj=logging_obj, - headers={}, - model=model, # Pass the deployment name - timeout=60.0, - ) - except Exception as e: - # If there's an error, we still want to check the mock calls - pass + response = await azure_chat_completion.aimage_generation( + data=data, + model_response=None, + azure_client_params=azure_client_params, + api_key=api_key, + input=[], + logging_obj=logging_obj, + headers={}, + model=model, # Pass the deployment name + timeout=60.0, + ) # Verify the mock was called assert mock_request.called, "HTTP request should have been made" From 56119742285265889eaf785ab688e265aeac8a76 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:17:07 +0530 Subject: [PATCH 45/50] Fix : litellm/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py --- litellm/anthropic_beta_headers_config.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/anthropic_beta_headers_config.json b/litellm/anthropic_beta_headers_config.json index 1e99e35aabe..4ebb5ddb609 100644 --- a/litellm/anthropic_beta_headers_config.json +++ b/litellm/anthropic_beta_headers_config.json @@ -24,7 +24,8 @@ "code-execution-2025-08-25", "skills-2025-10-02", "files-api-2025-04-14", - "fast-mode-2026-02-01" + "fast-mode-2026-02-01", + "mcp-servers-2025-12-04" ], "vertex_ai": [ "prompt-caching-scope-2026-01-05" From 0b5cb47c03ab590eb5f80e35695bc2071af318a7 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:34:11 +0530 Subject: [PATCH 46/50] fix: Missing return statement for async streaming --- litellm/llms/custom_httpx/llm_http_handler.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 3907ff7abf7..95db8ec64b3 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -438,7 +438,7 @@ class BaseLLMHTTPHandler: provider_config=provider_config, fake_stream=fake_stream, ) - response = self.acompletion_stream_function( + return self.acompletion_stream_function( model=model, messages=messages, api_base=api_base, From 5702cc7e13b8cd8122cb570c24b1adeae02c3890 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:38:03 +0530 Subject: [PATCH 47/50] Fix: get_supported_anthropic_messages_params --- .../experimental_pass_through/messages/transformation.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index 7d93e184090..043a70f3c67 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -46,12 +46,9 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "thinking", "context_management", "output_format", -<<<<<<< litellm_v1_messages_claude_4_6 "inference_geo", "speed", -======= "output_config", ->>>>>>> main # TODO: Add Anthropic `metadata` support # "metadata", ] From 125e11d36e46de24081da3d271a3fa8290d021cb Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:43:52 +0530 Subject: [PATCH 48/50] Fix mypy issues --- litellm/integrations/websearch_interception/handler.py | 7 ++++--- .../integrations/websearch_interception/transformation.py | 4 ++-- 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 7e8fa66c493..82d91d811ef 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -490,7 +490,8 @@ class WebSearchInterceptionLogger(CustomLogger): ) # Make follow-up request with search results - follow_up_messages = messages + [assistant_message, user_message] + # Type cast: user_message is a Dict for Anthropic format (default response_format) + follow_up_messages = messages + [assistant_message, cast(Dict, user_message)] verbose_logger.debug( "WebSearchInterception: Making follow-up request with search results" @@ -702,10 +703,10 @@ class WebSearchInterceptionLogger(CustomLogger): # Make follow-up request with search results # For OpenAI format, tool_messages_or_user is a list of tool messages if response_format == "openai": - follow_up_messages = messages + [assistant_message] + tool_messages_or_user + follow_up_messages = messages + [assistant_message] + cast(List[Dict], tool_messages_or_user) else: # For Anthropic format (shouldn't happen in this method, but handle it) - follow_up_messages = messages + [assistant_message, tool_messages_or_user] + follow_up_messages = messages + [assistant_message, cast(Dict, tool_messages_or_user)] verbose_logger.debug( "WebSearchInterception: Making follow-up chat completion request with search results" diff --git a/litellm/integrations/websearch_interception/transformation.py b/litellm/integrations/websearch_interception/transformation.py index ad92f5d03d0..e44ec35c3a2 100644 --- a/litellm/integrations/websearch_interception/transformation.py +++ b/litellm/integrations/websearch_interception/transformation.py @@ -4,7 +4,7 @@ WebSearch Tool Transformation Transforms between Anthropic/OpenAI tool_use format and LiteLLM search format. """ import json -from typing import Any, Dict, List, Tuple +from typing import Any, Dict, List, Tuple, Union from litellm._logging import verbose_logger from litellm.constants import LITELLM_WEB_SEARCH_TOOL_NAME @@ -224,7 +224,7 @@ class WebSearchTransformation: tool_calls: List[Dict], search_results: List[str], response_format: str = "anthropic", - ) -> Tuple[Dict, Dict]: + ) -> Tuple[Dict, Union[Dict, List[Dict]]]: """ Transform LiteLLM search results to Anthropic/OpenAI tool_result format. From 2d18ae4f9e92478d627973650dd95f69adec351e Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 9 Feb 2026 17:44:39 +0530 Subject: [PATCH 49/50] Fix mypy issues --- litellm/integrations/websearch_interception/handler.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 82d91d811ef..1277cac51d7 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -630,7 +630,7 @@ class WebSearchInterceptionLogger(CustomLogger): ) raise - async def _execute_chat_completion_agentic_loop( + async def _execute_chat_completion_agentic_loop( # noqa: PLR0915 self, model: str, messages: List[Dict], From 9532ad0fab15087a883a89cee1c6a264be6e2100 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 9 Feb 2026 10:03:43 -0800 Subject: [PATCH 50/50] docs fix (#20768) --- .../docs/proxy/forward_client_headers.md | 46 +++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/docs/my-website/docs/proxy/forward_client_headers.md b/docs/my-website/docs/proxy/forward_client_headers.md index 5477ffe87aa..2155a7517be 100644 --- a/docs/my-website/docs/proxy/forward_client_headers.md +++ b/docs/my-website/docs/proxy/forward_client_headers.md @@ -6,6 +6,52 @@ Control which model groups can forward client headers to the underlying LLM prov By default, LiteLLM does not forward client headers to LLM provider APIs for security reasons. However, you can selectively enable header forwarding for specific model groups using the `forward_client_headers_to_llm_api` setting. +## How it Works + +LiteLLM does **not** forward all client headers to the LLM provider. Instead, it uses an **allowlist** approach β€” only headers matching specific rules are forwarded. This ensures sensitive headers (like your LiteLLM API key) are never accidentally sent to upstream providers. + +```mermaid +sequenceDiagram + participant Client as Client (SDK / curl) + participant Proxy as LiteLLM Proxy + participant Filter as Header Filter (Allowlist) + participant LLM as LLM Provider (OpenAI, Anthropic, etc.) + + Client->>Proxy: Request with all headers
(Authorization, x-trace-id,
x-custom-header, anthropic-beta, etc.) + + Proxy->>Filter: Check forward_client_headers_to_llm_api
setting for this model group + + Note over Filter: Allowlist rules:
1. Headers starting with "x-" βœ…
2. "anthropic-beta" βœ…
3. "x-stainless-*" ❌ (blocked)
4. All other headers ❌ (blocked) + + Filter-->>Proxy: Return only allowed headers + + Proxy->>LLM: Request with filtered headers
(x-trace-id, x-custom-header,
anthropic-beta) + + LLM-->>Proxy: Response + Proxy-->>Client: Response +``` + +### Header Allowlist Rules + +The following rules determine which headers are forwarded (see [`_get_forwardable_headers`](https://github.com/litellm/litellm/blob/main/litellm/proxy/litellm_pre_call_utils.py) in `litellm/proxy/litellm_pre_call_utils.py`): + +| Rule | Example | Forwarded? | +|---|---|---| +| Headers starting with `x-` | `x-trace-id`, `x-custom-header`, `x-request-source` | βœ… Yes | +| `anthropic-beta` header | `anthropic-beta: prompt-caching-2024-07-31` | βœ… Yes | +| Headers starting with `x-stainless-*` | `x-stainless-lang`, `x-stainless-arch` | ❌ No (causes OpenAI SDK issues) | +| Standard HTTP headers | `Authorization`, `Content-Type`, `Host` | ❌ No | +| Other provider headers | `Accept`, `User-Agent` | ❌ No | + +### Additional Header Mechanisms + +| Mechanism | Description | Reference | +|---|---|---| +| **`x-pass-` prefix** | Headers prefixed with `x-pass-` are always forwarded with the prefix stripped, regardless of settings. E.g., `x-pass-anthropic-beta: value` β†’ `anthropic-beta: value`. Works for all pass-through endpoints. | [Source code](https://github.com/litellm/litellm/blob/main/litellm/passthrough/utils.py) | +| **`openai-organization`** | Forwarded only when `forward_openai_org_id: true` is set in `general_settings`. | [Forward OpenAI Org ID](#enable-globally) | +| **User information headers** | When `add_user_information_to_llm_headers: true`, LiteLLM adds `x-litellm-user-id`, `x-litellm-org-id`, etc. | [User Information Headers](#user-information-headers-optional) | +| **Vertex AI pass-through** | Uses a separate, stricter allowlist: only `anthropic-beta` and `content-type`. | [Source code](https://github.com/litellm/litellm/blob/main/litellm/constants.py) | + ## Configuration ## Enable Globally