diff --git a/litellm/llms/azure/chat/gpt_5_transformation.py b/litellm/llms/azure/chat/gpt_5_transformation.py index a70e008d66b..0292c73f998 100644 --- a/litellm/llms/azure/chat/gpt_5_transformation.py +++ b/litellm/llms/azure/chat/gpt_5_transformation.py @@ -69,7 +69,10 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config): # gpt-5.1/5.2/5.4 support reasoning_effort='none', but other gpt-5 models don't # See: https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning - is_gpt_5_1 = self.is_model_gpt_5_1_model(model) + model_for_check = model.replace(self.GPT5_SERIES_ROUTE, "") + is_gpt_5_1 = self._supports_reasoning_effort_level( + model_for_check, "none" + ) if reasoning_effort_value == "none" and not is_gpt_5_1: if litellm.drop_params is True or ( @@ -97,7 +100,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config): self, non_default_params=non_default_params, optional_params=optional_params, - model=model, + model=model_for_check, drop_params=drop_params, ) diff --git a/litellm/llms/openai/chat/gpt_5_transformation.py b/litellm/llms/openai/chat/gpt_5_transformation.py index beb76f3d80a..d357cffe664 100644 --- a/litellm/llms/openai/chat/gpt_5_transformation.py +++ b/litellm/llms/openai/chat/gpt_5_transformation.py @@ -58,6 +58,12 @@ class OpenAIGPT5Config(OpenAIGPTConfig): """Check if the model is specifically a GPT-5 Codex variant.""" return "gpt-5-codex" in model + @classmethod + def is_model_gpt_5_1_model(cls, model: str) -> bool: + """Check if the model is a gpt-5.1 variant (e.g. gpt-5.1, gpt-5.1-codex).""" + model_name = model.split("/")[-1] + return model_name.startswith("gpt-5.1") + @classmethod def is_model_gpt_5_2_model(cls, model: str) -> bool: """Check if the model is a gpt-5.2 variant (including pro).""" @@ -185,16 +191,33 @@ class OpenAIGPT5Config(OpenAIGPTConfig): "max_tokens" ) - # gpt-5.4: function calls not supported when reasoning_effort != "none" - # Drop reasoning_effort when tools are present (small minority of volume) + # gpt-5.4: function calls not supported when reasoning_effort != "none" in chat completions API + # However, the Responses API supports both tools and reasoning together + # So we keep reasoning_effort if the request will be routed to Responses API if self.is_model_gpt_5_4_model(model): has_tools = bool( non_default_params.get("tools") or optional_params.get("tools") ) if has_tools and reasoning_effort not in (None, "none"): - non_default_params.pop("reasoning_effort", None) - optional_params.pop("reasoning_effort", None) - reasoning_effort = None + # Check if this will be routed to Responses API + # If so, keep reasoning_effort; otherwise drop it for chat completions API + model_name = model.split("/")[-1] + will_route_to_responses = False + if model_name.startswith("gpt-5."): + try: + version_str = model_name.replace("gpt-5.", "").split("-")[0] + if "." in version_str: + major_version = int(version_str.split(".")[0]) + else: + major_version = int(version_str) + will_route_to_responses = major_version >= 4 + except (ValueError, IndexError): + pass + + if not will_route_to_responses: + non_default_params.pop("reasoning_effort", None) + optional_params.pop("reasoning_effort", None) + reasoning_effort = None # gpt-5.1/5.2 support logprobs, top_p, top_logprobs only when reasoning_effort="none" supports_none = self._supports_reasoning_effort_level(model, "none") diff --git a/litellm/main.py b/litellm/main.py index 52e7475169e..993b14d8970 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -927,8 +927,56 @@ def responses_api_bridge_check( model: str, custom_llm_provider: str, web_search_options: Optional[OpenAIWebSearchOptions] = None, + tools: Optional[List] = None, + reasoning_effort: Optional[str] = None, ) -> Tuple[dict, str]: + """ + Check if a chat completion request should be routed to the Responses API. + + Routes to Responses API when: + 1. Model name starts with "responses/" + 2. xAI provider with web_search_options + 3. OpenAI provider with GPT-5.4+ model, tools, and reasoning_effort (not "none") + + Args: + model: Model name (e.g., "gpt-5.4", "responses/grok-3") + custom_llm_provider: Provider name (e.g., "openai", "xai", "azure") + web_search_options: Optional web search configuration + tools: Optional list of tools/functions + reasoning_effort: Optional reasoning effort level + + Returns: + Tuple of (model_info dict, updated model name with "responses/" prefix removed) + """ model_info: Dict[str, Any] = {} + + def _check_gpt_54_or_above_routing() -> bool: + """Check if model is GPT-5.4+ and should route to responses API. + + For OpenAI GPT-5.4+ models, when both tools and reasoning are present, + route to Responses API which supports both features together. + The chat completion API would otherwise drop reasoning_effort when tools are present. + """ + if ( + custom_llm_provider == "openai" + and tools is not None + and len(tools) > 0 + and reasoning_effort is not None + and reasoning_effort != "none" + ): + model_name = model.split("/")[-1] + if model_name.startswith("gpt-5."): + try: + version_str = model_name.replace("gpt-5.", "").split("-")[0] + if "." in version_str: + major_version = int(version_str.split(".")[0]) + else: + major_version = int(version_str) + return major_version >= 4 + except (ValueError, IndexError): + pass + return False + try: model_info = cast( dict, @@ -944,6 +992,10 @@ def responses_api_bridge_check( if web_search_options is not None and custom_llm_provider == "xai": model_info["mode"] = "responses" model = model.replace("responses/", "") + + if _check_gpt_54_or_above_routing(): + model_info["mode"] = "responses" + model = model.replace("responses/", "") except Exception as e: verbose_logger.debug("Error getting model info: {}".format(e)) @@ -953,6 +1005,9 @@ def responses_api_bridge_check( model = model.replace("responses/", "") mode = "responses" model_info["mode"] = mode + elif _check_gpt_54_or_above_routing(): + model_info["mode"] = "responses" + model = model.replace("responses/", "") return model_info, model @@ -1569,6 +1624,8 @@ def completion( # type: ignore # noqa: PLR0915 model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, + tools=tools, + reasoning_effort=reasoning_effort, ) if model_info.get("mode") == "responses":