mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
Add rotuing to responses when tools + reasoning for gpt-5.4
This commit is contained in:
parent
814e36353b
commit
2aed1f5b15
3 changed files with 90 additions and 7 deletions
|
|
@ -69,7 +69,10 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
|
|||
|
||||
# gpt-5.1/5.2/5.4 support reasoning_effort='none', but other gpt-5 models don't
|
||||
# See: https://learn.microsoft.com/en-us/azure/ai-foundry/openai/how-to/reasoning
|
||||
is_gpt_5_1 = self.is_model_gpt_5_1_model(model)
|
||||
model_for_check = model.replace(self.GPT5_SERIES_ROUTE, "")
|
||||
is_gpt_5_1 = self._supports_reasoning_effort_level(
|
||||
model_for_check, "none"
|
||||
)
|
||||
|
||||
if reasoning_effort_value == "none" and not is_gpt_5_1:
|
||||
if litellm.drop_params is True or (
|
||||
|
|
@ -97,7 +100,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
|
|||
self,
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
model=model_for_check,
|
||||
drop_params=drop_params,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -58,6 +58,12 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
"""Check if the model is specifically a GPT-5 Codex variant."""
|
||||
return "gpt-5-codex" in model
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_1_model(cls, model: str) -> bool:
|
||||
"""Check if the model is a gpt-5.1 variant (e.g. gpt-5.1, gpt-5.1-codex)."""
|
||||
model_name = model.split("/")[-1]
|
||||
return model_name.startswith("gpt-5.1")
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_2_model(cls, model: str) -> bool:
|
||||
"""Check if the model is a gpt-5.2 variant (including pro)."""
|
||||
|
|
@ -185,16 +191,33 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
"max_tokens"
|
||||
)
|
||||
|
||||
# gpt-5.4: function calls not supported when reasoning_effort != "none"
|
||||
# Drop reasoning_effort when tools are present (small minority of volume)
|
||||
# gpt-5.4: function calls not supported when reasoning_effort != "none" in chat completions API
|
||||
# However, the Responses API supports both tools and reasoning together
|
||||
# So we keep reasoning_effort if the request will be routed to Responses API
|
||||
if self.is_model_gpt_5_4_model(model):
|
||||
has_tools = bool(
|
||||
non_default_params.get("tools") or optional_params.get("tools")
|
||||
)
|
||||
if has_tools and reasoning_effort not in (None, "none"):
|
||||
non_default_params.pop("reasoning_effort", None)
|
||||
optional_params.pop("reasoning_effort", None)
|
||||
reasoning_effort = None
|
||||
# Check if this will be routed to Responses API
|
||||
# If so, keep reasoning_effort; otherwise drop it for chat completions API
|
||||
model_name = model.split("/")[-1]
|
||||
will_route_to_responses = False
|
||||
if model_name.startswith("gpt-5."):
|
||||
try:
|
||||
version_str = model_name.replace("gpt-5.", "").split("-")[0]
|
||||
if "." in version_str:
|
||||
major_version = int(version_str.split(".")[0])
|
||||
else:
|
||||
major_version = int(version_str)
|
||||
will_route_to_responses = major_version >= 4
|
||||
except (ValueError, IndexError):
|
||||
pass
|
||||
|
||||
if not will_route_to_responses:
|
||||
non_default_params.pop("reasoning_effort", None)
|
||||
optional_params.pop("reasoning_effort", None)
|
||||
reasoning_effort = None
|
||||
|
||||
# gpt-5.1/5.2 support logprobs, top_p, top_logprobs only when reasoning_effort="none"
|
||||
supports_none = self._supports_reasoning_effort_level(model, "none")
|
||||
|
|
|
|||
|
|
@ -927,8 +927,56 @@ def responses_api_bridge_check(
|
|||
model: str,
|
||||
custom_llm_provider: str,
|
||||
web_search_options: Optional[OpenAIWebSearchOptions] = None,
|
||||
tools: Optional[List] = None,
|
||||
reasoning_effort: Optional[str] = None,
|
||||
) -> Tuple[dict, str]:
|
||||
"""
|
||||
Check if a chat completion request should be routed to the Responses API.
|
||||
|
||||
Routes to Responses API when:
|
||||
1. Model name starts with "responses/"
|
||||
2. xAI provider with web_search_options
|
||||
3. OpenAI provider with GPT-5.4+ model, tools, and reasoning_effort (not "none")
|
||||
|
||||
Args:
|
||||
model: Model name (e.g., "gpt-5.4", "responses/grok-3")
|
||||
custom_llm_provider: Provider name (e.g., "openai", "xai", "azure")
|
||||
web_search_options: Optional web search configuration
|
||||
tools: Optional list of tools/functions
|
||||
reasoning_effort: Optional reasoning effort level
|
||||
|
||||
Returns:
|
||||
Tuple of (model_info dict, updated model name with "responses/" prefix removed)
|
||||
"""
|
||||
model_info: Dict[str, Any] = {}
|
||||
|
||||
def _check_gpt_54_or_above_routing() -> bool:
|
||||
"""Check if model is GPT-5.4+ and should route to responses API.
|
||||
|
||||
For OpenAI GPT-5.4+ models, when both tools and reasoning are present,
|
||||
route to Responses API which supports both features together.
|
||||
The chat completion API would otherwise drop reasoning_effort when tools are present.
|
||||
"""
|
||||
if (
|
||||
custom_llm_provider == "openai"
|
||||
and tools is not None
|
||||
and len(tools) > 0
|
||||
and reasoning_effort is not None
|
||||
and reasoning_effort != "none"
|
||||
):
|
||||
model_name = model.split("/")[-1]
|
||||
if model_name.startswith("gpt-5."):
|
||||
try:
|
||||
version_str = model_name.replace("gpt-5.", "").split("-")[0]
|
||||
if "." in version_str:
|
||||
major_version = int(version_str.split(".")[0])
|
||||
else:
|
||||
major_version = int(version_str)
|
||||
return major_version >= 4
|
||||
except (ValueError, IndexError):
|
||||
pass
|
||||
return False
|
||||
|
||||
try:
|
||||
model_info = cast(
|
||||
dict,
|
||||
|
|
@ -944,6 +992,10 @@ def responses_api_bridge_check(
|
|||
if web_search_options is not None and custom_llm_provider == "xai":
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
|
||||
if _check_gpt_54_or_above_routing():
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
except Exception as e:
|
||||
verbose_logger.debug("Error getting model info: {}".format(e))
|
||||
|
||||
|
|
@ -953,6 +1005,9 @@ def responses_api_bridge_check(
|
|||
model = model.replace("responses/", "")
|
||||
mode = "responses"
|
||||
model_info["mode"] = mode
|
||||
elif _check_gpt_54_or_above_routing():
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
return model_info, model
|
||||
|
||||
|
||||
|
|
@ -1569,6 +1624,8 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
web_search_options=web_search_options,
|
||||
tools=tools,
|
||||
reasoning_effort=reasoning_effort,
|
||||
)
|
||||
|
||||
if model_info.get("mode") == "responses":
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue