fix(litellm): gate the gpt-5.4+ responses bridge on function tools specifically

OpenAI's chat completions rejection applies to function tools only;
custom (grammar) tools are served natively with reasoning on, live-proven
by a 200 on a custom-only gpt-5.6 chat request. Gating on any truthy
tools needlessly bridged custom-only requests, and the bridge maps custom
tool calls back function-shaped, so the native chat custom tool_call
surface added earlier in this PR was bypassed exactly where chat serves
it natively. The gate now checks for a function-type tool in either the
nested chat or flat Responses def shape; the same coarseness existed on
the explicit-effort arm before this PR and is fixed by the shared leg
This commit is contained in:
Tin Chi Lo 2026-07-21 18:10:16 -07:00
parent 6d102ea559
commit 7276f44b1d
2 changed files with 73 additions and 6 deletions

View file

@ -1022,14 +1022,20 @@ def responses_api_bridge_check(
# ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects
# those keys.
#
# - gpt-5.4+: function tools with reasoning active must be bridged. OpenAI enables
# - gpt-5.4+: FUNCTION tools with reasoning active must be bridged. OpenAI enables
# reasoning by default for these models (unset reasoning_effort means medium
# server-side), and Chat Completions rejects tools whenever reasoning is on
# ("Function tools with reasoning_effort are not supported ... use /v1/responses
# or set reasoning_effort to 'none'"), so only an explicit ``"none"`` keeps the
# request chat-servable.
# server-side), and Chat Completions rejects function tools whenever reasoning is
# on ("Function tools with reasoning_effort are not supported ... use
# /v1/responses or set reasoning_effort to 'none'"), so only an explicit
# ``"none"`` keeps the request chat-servable. Custom (grammar) tools are served
# natively by Chat Completions with reasoning on, so custom-only requests stay on
# chat and keep their native custom tool_call response shape.
# - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning
# summary alias is present with ``reasoning_effort`` (tools alone stay on chat).
has_function_tool = any(
(tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function")
for tool in (tools or [])
)
if (
custom_llm_provider in ("openai", "azure")
and model_info.get("mode") != "responses"
@ -1037,7 +1043,9 @@ def responses_api_bridge_check(
and not OpenAIGPT5Config.is_model_gpt_5_search_model(model)
and (
(reasoning_effort is not None and reasoning_summary is not None)
or (OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and tools and reasoning_effort != "none")
or (
OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model) and has_function_tool and reasoning_effort != "none"
)
)
):
model_info["mode"] = "responses"

View file

@ -889,6 +889,65 @@ def test_responses_api_bridge_check_reasoning_none_with_summary_still_routes_to_
assert model_info.get("mode") == "responses"
def test_responses_api_bridge_check_gpt_5_4_custom_tools_only_stays_chat():
"""
Chat Completions serves custom (grammar) tools natively with reasoning on; only
FUNCTION tools trigger the OpenAI rejection. Custom-only requests must stay on chat
so responses keep the native custom tool_call shape instead of the bridge's
function-shaped mapping.
"""
from litellm.main import responses_api_bridge_check
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
mock_get_model_info.return_value = {"max_tokens": 128000}
model_info, model = responses_api_bridge_check(
model="gpt-5.6",
custom_llm_provider="openai",
tools=[{"type": "custom", "custom": {"name": "ApplyPatch", "description": "V4A patch"}}],
reasoning_effort=None,
)
assert model == "gpt-5.6"
assert model_info.get("mode") != "responses"
def test_responses_api_bridge_check_gpt_5_4_mixed_function_and_custom_tools_routes_to_responses():
"""One function tool in the mix is enough to make chat unservable with reasoning on."""
from litellm.main import responses_api_bridge_check
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
mock_get_model_info.return_value = {"max_tokens": 128000}
model_info, model = responses_api_bridge_check(
model="gpt-5.6",
custom_llm_provider="openai",
tools=[
{"type": "custom", "custom": {"name": "ApplyPatch"}},
{"type": "function", "function": {"name": "shell"}},
],
reasoning_effort=None,
)
assert model == "gpt-5.6"
assert model_info.get("mode") == "responses"
def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_responses():
"""Responses-style flat function tool defs still count as function tools."""
from litellm.main import responses_api_bridge_check
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
mock_get_model_info.return_value = {"max_tokens": 128000}
model_info, model = responses_api_bridge_check(
model="gpt-5.6",
custom_llm_provider="openai",
tools=[{"type": "function", "name": "shell", "parameters": {"type": "object"}}],
reasoning_effort=None,
)
assert model == "gpt-5.6"
assert model_info.get("mode") == "responses"
def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat():
"""Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge."""
from litellm.main import responses_api_bridge_check