Add rotuing to responses when tools + reasoning for gpt-5.4

This commit is contained in:
Sameer Kankute 2026-03-09 01:00:17 +05:30
parent 814e36353b
commit 2aed1f5b15
3 changed files with 90 additions and 7 deletions

View file

@ -69,7 +69,10 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
# gpt-5.1/5.2/5.4 support reasoning_effort='none', but other gpt-5 models don't
# See: https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning
is_gpt_5_1 = self.is_model_gpt_5_1_model(model)
model_for_check = model.replace(self.GPT5_SERIES_ROUTE, "")
is_gpt_5_1 = self._supports_reasoning_effort_level(
model_for_check, "none"
)
if reasoning_effort_value == "none" and not is_gpt_5_1:
if litellm.drop_params is True or (
@ -97,7 +100,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
self,
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
model=model_for_check,
drop_params=drop_params,
)

View file

@ -58,6 +58,12 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
"""Check if the model is specifically a GPT-5 Codex variant."""
return "gpt-5-codex" in model
@classmethod
def is_model_gpt_5_1_model(cls, model: str) -> bool:
"""Check if the model is a gpt-5.1 variant (e.g. gpt-5.1, gpt-5.1-codex)."""
model_name = model.split("/")[-1]
return model_name.startswith("gpt-5.1")
@classmethod
def is_model_gpt_5_2_model(cls, model: str) -> bool:
"""Check if the model is a gpt-5.2 variant (including pro)."""
@ -185,16 +191,33 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
"max_tokens"
)
# gpt-5.4: function calls not supported when reasoning_effort != "none"
# Drop reasoning_effort when tools are present (small minority of volume)
# gpt-5.4: function calls not supported when reasoning_effort != "none" in chat completions API
# However, the Responses API supports both tools and reasoning together
# So we keep reasoning_effort if the request will be routed to Responses API
if self.is_model_gpt_5_4_model(model):
has_tools = bool(
non_default_params.get("tools") or optional_params.get("tools")
)
if has_tools and reasoning_effort not in (None, "none"):
non_default_params.pop("reasoning_effort", None)
optional_params.pop("reasoning_effort", None)
reasoning_effort = None
# Check if this will be routed to Responses API
# If so, keep reasoning_effort; otherwise drop it for chat completions API
model_name = model.split("/")[-1]
will_route_to_responses = False
if model_name.startswith("gpt-5."):
try:
version_str = model_name.replace("gpt-5.", "").split("-")[0]
if "." in version_str:
major_version = int(version_str.split(".")[0])
else:
major_version = int(version_str)
will_route_to_responses = major_version >= 4
except (ValueError, IndexError):
pass
if not will_route_to_responses:
non_default_params.pop("reasoning_effort", None)
optional_params.pop("reasoning_effort", None)
reasoning_effort = None
# gpt-5.1/5.2 support logprobs, top_p, top_logprobs only when reasoning_effort="none"
supports_none = self._supports_reasoning_effort_level(model, "none")

View file

@ -927,8 +927,56 @@ def responses_api_bridge_check(
model: str,
custom_llm_provider: str,
web_search_options: Optional[OpenAIWebSearchOptions] = None,
tools: Optional[List] = None,
reasoning_effort: Optional[str] = None,
) -> Tuple[dict, str]:
"""
Check if a chat completion request should be routed to the Responses API.
Routes to Responses API when:
1. Model name starts with "responses/"
2. xAI provider with web_search_options
3. OpenAI provider with GPT-5.4+ model, tools, and reasoning_effort (not "none")
Args:
model: Model name (e.g., "gpt-5.4", "responses/grok-3")
custom_llm_provider: Provider name (e.g., "openai", "xai", "azure")
web_search_options: Optional web search configuration
tools: Optional list of tools/functions
reasoning_effort: Optional reasoning effort level
Returns:
Tuple of (model_info dict, updated model name with "responses/" prefix removed)
"""
model_info: Dict[str, Any] = {}
def _check_gpt_54_or_above_routing() -> bool:
"""Check if model is GPT-5.4+ and should route to responses API.
For OpenAI GPT-5.4+ models, when both tools and reasoning are present,
route to Responses API which supports both features together.
The chat completion API would otherwise drop reasoning_effort when tools are present.
"""
if (
custom_llm_provider == "openai"
and tools is not None
and len(tools) > 0
and reasoning_effort is not None
and reasoning_effort != "none"
):
model_name = model.split("/")[-1]
if model_name.startswith("gpt-5."):
try:
version_str = model_name.replace("gpt-5.", "").split("-")[0]
if "." in version_str:
major_version = int(version_str.split(".")[0])
else:
major_version = int(version_str)
return major_version >= 4
except (ValueError, IndexError):
pass
return False
try:
model_info = cast(
dict,
@ -944,6 +992,10 @@ def responses_api_bridge_check(
if web_search_options is not None and custom_llm_provider == "xai":
model_info["mode"] = "responses"
model = model.replace("responses/", "")
if _check_gpt_54_or_above_routing():
model_info["mode"] = "responses"
model = model.replace("responses/", "")
except Exception as e:
verbose_logger.debug("Error getting model info: {}".format(e))
@ -953,6 +1005,9 @@ def responses_api_bridge_check(
model = model.replace("responses/", "")
mode = "responses"
model_info["mode"] = mode
elif _check_gpt_54_or_above_routing():
model_info["mode"] = "responses"
model = model.replace("responses/", "")
return model_info, model
@ -1569,6 +1624,8 @@ def completion( # type: ignore # noqa: PLR0915
model=model,
custom_llm_provider=custom_llm_provider,
web_search_options=web_search_options,
tools=tools,
reasoning_effort=reasoning_effort,
)
if model_info.get("mode") == "responses":