From d516a72c0587fe813d7fdce39e5741fd74f2f660 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 21:47:51 -0700 Subject: [PATCH] fix(litellm): scope the unset-effort responses bridge to constraint-enforcing endpoints Chat-only OpenAI-compatible backends registered under the openai provider with custom api_base and gpt-5.4+ model names served tools-without-reasoning fine and have no /responses route, so the unset-effort arm added for real OpenAI would have silently rerouted previously working deployments. The arm now fires only when api_base is unset (default OpenAI endpoint) or the provider is azure; an explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base. Flagged lines also modernized to PEP 604 --- .../convert_dict_to_response.py | 9 +-- .../llms/openai/chat/gpt_transformation.py | 4 +- litellm/main.py | 17 +++++- tests/test_litellm/test_main.py | 58 +++++++++++++++++++ 4 files changed, 78 insertions(+), 10 deletions(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index c5cfdea9ffe..1b23db87264 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -531,12 +531,9 @@ class LiteLLMResponseObjectHandler: def _should_convert_tool_call_to_json_mode( - tool_calls: Optional[ - Union[ - List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]], - List[DatabricksTool], - ] - ] = None, + tool_calls: ( + list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | list[DatabricksTool] | None + ) = None, convert_tool_call_to_json_mode: Optional[bool] = None, ) -> bool: """ diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index 129a9b51d0d..e4492a8aba6 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -533,9 +533,7 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): for choice in choices: ## HANDLE JSON MODE - anthropic returns single function call] tool_calls = choice["message"].get("tool_calls", None) - new_tool_calls: Optional[ - List[Union[ChatCompletionMessageToolCall, ChatCompletionMessageCustomToolCall]] - ] = None + new_tool_calls: list[ChatCompletionMessageToolCall | ChatCompletionMessageCustomToolCall] | None = None message_content = choice["message"].get("content", None) if tool_calls is not None: _openai_tool_calls = [] diff --git a/litellm/main.py b/litellm/main.py index b6c6b44a6f7..d008d976130 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -986,6 +986,7 @@ def responses_api_bridge_check( tools: Optional[List[Any]] = None, reasoning_effort: Optional[Any] = None, reasoning_summary: Optional[Any] = None, + api_base: str | None = None, ) -> Tuple[dict, str]: model_info: Dict[str, Any] = {} @@ -1030,6 +1031,12 @@ def responses_api_bridge_check( # ``"none"`` keeps the request chat-servable. Custom (grammar) tools are served # natively by Chat Completions with reasoning on, so custom-only requests stay on # chat and keep their native custom tool_call response shape. + # - The UNSET-effort arm only fires against endpoints known to enforce that + # constraint (the default OpenAI endpoint, or Azure OpenAI where api_base is + # always set): chat-only OpenAI-compatible backends registered under the openai + # provider with a custom api_base and gpt-5.4+ model names serve tools without + # reasoning fine and have no /responses route, so they keep pre-existing + # behavior (bridge only on an explicit reasoning_effort). # - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning # summary alias is present with ``reasoning_effort`` (tools alone stay on chat). has_function_tool = any( @@ -1040,6 +1047,7 @@ def responses_api_bridge_check( reasoning_active = reasoning_effort.get("effort") != "none" or reasoning_effort.get("summary") is not None else: reasoning_active = reasoning_effort != "none" + on_constraint_enforcing_endpoint = custom_llm_provider == "azure" or api_base is None if ( custom_llm_provider in ("openai", "azure") and model_info.get("mode") != "responses" @@ -1047,7 +1055,12 @@ def responses_api_bridge_check( and not OpenAIGPT5Config.is_model_gpt_5_search_model(model) and ( (reasoning_effort is not None and reasoning_summary is not None) - or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_active) + or ( + OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) + and has_function_tool + and reasoning_active + and (reasoning_effort is not None or on_constraint_enforcing_endpoint) + ) ) ): model_info["mode"] = "responses" @@ -5173,6 +5186,7 @@ def completion( # type: ignore model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, + api_base=api_base, ) if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model): @@ -5412,6 +5426,7 @@ def completion( # type: ignore tools=tools, reasoning_effort=reasoning_effort, reasoning_summary=_reasoning_summary_for_bridge, + api_base=api_base, ) # Use base_model (the true underlying model) for Azure model-type diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 60558760f8e..f72d2b5e23b 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -998,6 +998,64 @@ def test_responses_api_bridge_check_dict_effort_none_with_summary_routes_to_resp assert model_info.get("mode") == "responses" +def test_responses_api_bridge_check_custom_api_base_with_unset_effort_stays_chat(): + """ + Chat-only OpenAI-compatible backends registered under the openai provider with a + custom api_base and gpt-5.4+ model names serve tools-without-reasoning fine and + have no /responses route; the unset-effort arm must not reroute them. + """ + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base="http://vllm.internal:8000/v1", + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") != "responses" + + +def test_responses_api_bridge_check_custom_api_base_with_explicit_effort_still_routes(): + """Explicit reasoning_effort keeps its pre-existing bridging behavior on any api_base.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort="high", + api_base="http://vllm.internal:8000/v1", + ) + + assert model == "gpt-5.6" + assert model_info.get("mode") == "responses" + + +def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes(): + """Azure OpenAI always sets api_base and does enforce the constraint; keep bridging.""" + from litellm.main import responses_api_bridge_check + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = {"max_tokens": 128000} + model_info, model = responses_api_bridge_check( + model="gpt-5.4", + custom_llm_provider="azure", + tools=[{"type": "function", "function": {"name": "get_capital"}}], + reasoning_effort=None, + api_base="https://myresource.openai.azure.com", + ) + + assert model == "gpt-5.4" + assert model_info.get("mode") == "responses" + + def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat(): """Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge.""" from litellm.main import responses_api_bridge_check