mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-27 01:22:18 +00:00
Merge pull request #42041 from BerriAI/litellm_azure_ai_gpt5_tools_responses_bridge
fix(azure_ai): bridge gpt-5.4+ function tools with reasoning to the Foundry Responses API
This commit is contained in:
commit
e7e4df9098
7 changed files with 301 additions and 26 deletions
|
|
@ -1115,7 +1115,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
responses_tools: Final[list[ALL_RESPONSES_API_TOOL_PARAMS]] = []
|
||||
for tool in tools:
|
||||
# convert function tool from chat completion to responses API format
|
||||
if tool.get("type") == "function":
|
||||
if tool.get("type") == "function" and isinstance(tool.get("function"), dict):
|
||||
function_tool = cast(ChatCompletionToolParamFunctionChunk, tool.get("function"))
|
||||
responses_tools.append(
|
||||
FunctionToolParam(
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ from urllib.parse import urlparse
|
|||
|
||||
import litellm
|
||||
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
|
||||
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
|
@ -150,6 +151,14 @@ def azure_ai_supports_native_responses(model: str | None, api_base: str | None)
|
|||
return AzureFoundryModelInfo.get_azure_ai_route(model) == "default"
|
||||
|
||||
|
||||
def foundry_chat_rejects_function_tools_while_reasoning(
|
||||
model: str, reasoning_effort: str | Mapping[str, object] | None
|
||||
) -> bool:
|
||||
if reasoning_effort is None:
|
||||
return OpenAIGPT5Config.is_model_gpt_6_plus_model(model)
|
||||
return OpenAIGPT5Config.is_model_gpt_5_6_plus_model(model)
|
||||
|
||||
|
||||
class AzureFoundryModelInfo(BaseLLMModelInfo):
|
||||
"""Model info for Azure AI / Azure Foundry models."""
|
||||
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
"""Support for OpenAI gpt-5 model family."""
|
||||
|
||||
import re
|
||||
from typing import Final
|
||||
|
||||
import litellm
|
||||
|
|
@ -11,6 +12,8 @@ from litellm.utils import (
|
|||
|
||||
from .gpt_transformation import OpenAIGPTConfig
|
||||
|
||||
_GPT_SERIES_VERSION: Final = re.compile(r"^gpt-(\d+)(?:\.(\d+))?(?=[.-]|$)")
|
||||
|
||||
|
||||
def _catalogue_declares_default_effort() -> bool:
|
||||
"""Whether the loaded cost map carries default_reasoning_effort for ANY entry.
|
||||
|
|
@ -112,20 +115,28 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
model_name: Final = model.split("/")[-1]
|
||||
return model_name.startswith("gpt-5.4")
|
||||
|
||||
@staticmethod
|
||||
def _gpt_series_version(model: str) -> tuple[int, int] | None:
|
||||
match: Final = _GPT_SERIES_VERSION.match(model.split("/")[-1])
|
||||
if match is None:
|
||||
return None
|
||||
return int(match.group(1)), int(match.group(2) or 0)
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_4_plus_model(cls, model: str) -> bool:
|
||||
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, etc., including pro)."""
|
||||
model_name: Final = model.split("/")[-1]
|
||||
if model_name.startswith("gpt-6"):
|
||||
return True
|
||||
if not model_name.startswith("gpt-5."):
|
||||
return False
|
||||
try:
|
||||
version_str: Final = model_name.replace("gpt-5.", "").split("-")[0]
|
||||
major: Final = version_str.split(".")[0]
|
||||
return int(major) >= 4
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
version: Final = cls._gpt_series_version(model)
|
||||
return version is not None and version >= (5, 4)
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_6_plus_model(cls, model: str) -> bool:
|
||||
version: Final = cls._gpt_series_version(model)
|
||||
return version is not None and version >= (5, 6)
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_6_plus_model(cls, model: str) -> bool:
|
||||
version: Final = cls._gpt_series_version(model)
|
||||
return version is not None and version >= (6, 0)
|
||||
|
||||
@classmethod
|
||||
def _model_map_lookup_name(cls, model: str) -> str:
|
||||
|
|
|
|||
|
|
@ -100,6 +100,10 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
|||
from litellm.litellm_core_utils.request_timeout_resolver import (
|
||||
get_configured_request_timeout,
|
||||
)
|
||||
from litellm.llms.azure_ai.common_utils import (
|
||||
azure_ai_supports_native_responses,
|
||||
foundry_chat_rejects_function_tools_while_reasoning,
|
||||
)
|
||||
from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig
|
||||
from litellm.llms.base_llm.base_model_iterator import (
|
||||
convert_model_response_to_streaming,
|
||||
|
|
@ -1106,10 +1110,18 @@ def responses_api_bridge_check(
|
|||
# provider with a custom api_base and gpt-5.4+ model names serve tools without
|
||||
# reasoning fine and have no /responses route, so they keep pre-existing
|
||||
# behavior (bridge only on an explicit reasoning_effort).
|
||||
# - Azure AI Foundry's OpenAI v1 hosts (azure_ai provider) enforce it later in the series:
|
||||
# an explicit effort with function tools is rejected from gpt-5.6 on, and the unset
|
||||
# effort only from gpt-6 on (gpt-5.6 serves tools with reasoning silently off), so the
|
||||
# azure_ai gate keys on those measured boundaries instead of gpt-5.4+.
|
||||
# - Older GPT-5 names (e.g. ``gpt-5``, ``gpt-5.1``): bridge only when a reasoning
|
||||
# summary alias is present with ``reasoning_effort`` (tools alone stay on chat).
|
||||
has_function_tool: Final = any(
|
||||
(tool.get("type") == "function" if isinstance(tool, dict) else getattr(tool, "type", None) == "function")
|
||||
(
|
||||
tool.get("type") == "function" and (isinstance(tool.get("function"), dict) or "name" in tool)
|
||||
if isinstance(tool, dict)
|
||||
else getattr(tool, "type", None) == "function"
|
||||
)
|
||||
for tool in (tools or ())
|
||||
)
|
||||
if isinstance(reasoning_effort, dict):
|
||||
|
|
@ -1118,28 +1130,35 @@ def responses_api_bridge_check(
|
|||
reasoning_active = reasoning_effort != "none"
|
||||
# The reasoning+tools constraint is enforced by the real OpenAI backend behind any api.openai.com
|
||||
# host (the default URL or a PrivateLink hostname such as <region>.privatelink.api.openai.com) and
|
||||
# by Azure OpenAI. Resolve the effective base arg>global>env>default exactly as the chat handler
|
||||
# does, so a custom base set via litellm.api_base or OPENAI_BASE_URL/OPENAI_API_BASE isn't misread
|
||||
# as the default and bridged to a /responses route it lacks. A whitespace-only base collapses to
|
||||
# the default too.
|
||||
# by Azure OpenAI through the azure provider. Resolve the effective OpenAI base arg>global>env>default
|
||||
# exactly as the chat handler does, so a custom base set via litellm.api_base or
|
||||
# OPENAI_BASE_URL/OPENAI_API_BASE isn't misread as the default and bridged to a /responses route it
|
||||
# lacks. A whitespace-only base collapses to the default too.
|
||||
resolved_api_base: Final = _resolve_openai_api_base(api_base).strip()
|
||||
on_foundry_openai_endpoint: Final = custom_llm_provider == "azure_ai" and azure_ai_supports_native_responses(
|
||||
model, api_base
|
||||
)
|
||||
on_constraint_enforcing_endpoint: Final = (
|
||||
custom_llm_provider == "azure" or resolved_api_base == "" or _is_openai_backed_api_base(resolved_api_base)
|
||||
)
|
||||
if (
|
||||
custom_llm_provider in ("openai", "azure")
|
||||
and model_info.get("mode") != "responses"
|
||||
and OpenAIGPT5Config.is_model_gpt_5_model(model)
|
||||
and not OpenAIGPT5Config.is_model_gpt_5_search_model(model)
|
||||
chat_rejects_function_tools: Final = (
|
||||
has_function_tool
|
||||
and reasoning_active
|
||||
and (
|
||||
(reasoning_effort is not None and reasoning_summary is not None)
|
||||
or (
|
||||
foundry_chat_rejects_function_tools_while_reasoning(model, reasoning_effort)
|
||||
if on_foundry_openai_endpoint
|
||||
else (
|
||||
OpenAIGPT5Config.is_model_gpt_5_4_plus_model(model)
|
||||
and has_function_tool
|
||||
and reasoning_active
|
||||
and (reasoning_effort is not None or on_constraint_enforcing_endpoint)
|
||||
)
|
||||
)
|
||||
)
|
||||
if (
|
||||
(custom_llm_provider in ("openai", "azure") or on_foundry_openai_endpoint)
|
||||
and model_info.get("mode") != "responses"
|
||||
and OpenAIGPT5Config.is_model_gpt_5_model(model)
|
||||
and not OpenAIGPT5Config.is_model_gpt_5_search_model(model)
|
||||
and ((reasoning_effort is not None and reasoning_summary is not None) or chat_rejects_function_tools)
|
||||
):
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
|
|
|
|||
|
|
@ -830,6 +830,24 @@ def test_convert_tools_to_responses_format():
|
|||
assert result[0]["name"] == "test"
|
||||
|
||||
|
||||
def test_convert_tools_to_responses_format_passes_flat_function_tool_through():
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
flat_tool = {
|
||||
"type": "function",
|
||||
"name": "shell",
|
||||
"description": "Run a shell command",
|
||||
"parameters": {"type": "object", "properties": {"cmd": {"type": "string"}}, "required": ["cmd"]},
|
||||
}
|
||||
|
||||
converted = handler._convert_tools_to_responses_format([flat_tool])
|
||||
|
||||
assert converted == [flat_tool]
|
||||
|
||||
|
||||
def test_extract_extra_body_params_reasoning_effort_override():
|
||||
"""Test that reasoning_effort from extra_body overrides top-level reasoning_effort"""
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
|
|
|
|||
|
|
@ -159,6 +159,58 @@ class TestOpenAIGPT5ConfigIsModelGpt54PlusModel:
|
|||
), f"Expected '{model}' NOT to be classified as gpt-5.4-or-newer"
|
||||
|
||||
|
||||
GPT5_6_PLUS_MODELS = [
|
||||
"gpt-6-astra",
|
||||
"openai/gpt-6-astra",
|
||||
"gpt-5.6",
|
||||
"gpt-5.6-sol",
|
||||
"gpt-5.6-terra",
|
||||
"gpt-5.10-preview",
|
||||
]
|
||||
|
||||
GPT5_PRE_5_6_MODELS = [
|
||||
"gpt-5",
|
||||
"gpt-5.4",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.5",
|
||||
"gpt-5.5-pro",
|
||||
"gpt-4o",
|
||||
]
|
||||
|
||||
GPT6_PLUS_MODELS = [
|
||||
"gpt-6-astra",
|
||||
"openai/gpt-6-astra",
|
||||
"gpt-6",
|
||||
"gpt-6.1-preview",
|
||||
]
|
||||
|
||||
GPT_PRE_6_MODELS = [
|
||||
"gpt-5.6-sol",
|
||||
"gpt-5.5",
|
||||
"gpt-5",
|
||||
"gpt-4o",
|
||||
]
|
||||
|
||||
|
||||
class TestOpenAIGPT5ConfigSeriesBoundaries:
|
||||
|
||||
@pytest.mark.parametrize("model", GPT5_6_PLUS_MODELS)
|
||||
def test_gpt5_6_plus_models_are_classified_as_5_6_plus(self, model: str):
|
||||
assert OpenAIGPT5Config.is_model_gpt_5_6_plus_model(model)
|
||||
|
||||
@pytest.mark.parametrize("model", GPT5_PRE_5_6_MODELS)
|
||||
def test_pre_5_6_models_are_not_classified_as_5_6_plus(self, model: str):
|
||||
assert not OpenAIGPT5Config.is_model_gpt_5_6_plus_model(model)
|
||||
|
||||
@pytest.mark.parametrize("model", GPT6_PLUS_MODELS)
|
||||
def test_gpt6_plus_models_are_classified_as_6_plus(self, model: str):
|
||||
assert OpenAIGPT5Config.is_model_gpt_6_plus_model(model)
|
||||
|
||||
@pytest.mark.parametrize("model", GPT_PRE_6_MODELS)
|
||||
def test_pre_6_models_are_not_classified_as_6_plus(self, model: str):
|
||||
assert not OpenAIGPT5Config.is_model_gpt_6_plus_model(model)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# AzureOpenAIGPT5Config
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
|
|||
|
|
@ -1049,6 +1049,35 @@ def test_responses_api_bridge_check_gpt_5_4_flat_function_tool_routes_to_respons
|
|||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"custom_llm_provider, model_name, api_base",
|
||||
[
|
||||
pytest.param("openai", "gpt-5.6", None, id="openai"),
|
||||
pytest.param("azure_ai", "gpt-6-astra", "https://myproject.services.ai.azure.com", id="azure-ai-foundry"),
|
||||
],
|
||||
)
|
||||
def test_responses_api_bridge_check_function_tool_without_body_stays_chat(
|
||||
monkeypatch, custom_llm_provider, model_name, api_base
|
||||
):
|
||||
import litellm
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model=model_name,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
tools=[{"type": "function"}],
|
||||
reasoning_effort=None,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert model == model_name
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_dict_effort_none_stays_chat():
|
||||
"""The escape hatch must honor litellm's dict form: {"effort": "none"} means reasoning off."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
|
@ -1308,6 +1337,68 @@ def test_responses_api_bridge_check_azure_with_api_base_and_unset_effort_routes(
|
|||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
_FOUNDRY_API_BASE: Final = "https://myproject.services.ai.azure.com"
|
||||
_FOUNDRY_FUNCTION_TOOL: Final = ({"type": "function", "function": {"name": "get_weather"}},)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name, api_base, reasoning_effort",
|
||||
[
|
||||
pytest.param("gpt-6-astra", _FOUNDRY_API_BASE, None, id="gpt-6-unset-effort"),
|
||||
pytest.param("gpt-6-astra", _FOUNDRY_API_BASE, "low", id="gpt-6-explicit-effort"),
|
||||
pytest.param("gpt-6-astra", "https://myresource.openai.azure.com", None, id="gpt-6-azure-openai-host"),
|
||||
pytest.param("gpt-5.6-sol", _FOUNDRY_API_BASE, "low", id="gpt-5.6-explicit-effort"),
|
||||
pytest.param("gpt-5.6-sol", _FOUNDRY_API_BASE, {"effort": "high"}, id="gpt-5.6-explicit-effort-dict"),
|
||||
],
|
||||
)
|
||||
def test_responses_api_bridge_check_azure_ai_foundry_rejected_tools_route_to_responses(
|
||||
model_name, api_base, reasoning_effort
|
||||
):
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model=model_name,
|
||||
custom_llm_provider="azure_ai",
|
||||
tools=_FOUNDRY_FUNCTION_TOOL,
|
||||
reasoning_effort=reasoning_effort,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert model == model_name
|
||||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name, api_base, reasoning_effort",
|
||||
[
|
||||
pytest.param("gpt-6-astra", _FOUNDRY_API_BASE, "none", id="explicit-none-stays-chat"),
|
||||
pytest.param("gpt-5.6-sol", _FOUNDRY_API_BASE, None, id="gpt-5.6-unset-effort-stays-chat"),
|
||||
pytest.param("gpt-5.6-sol", _FOUNDRY_API_BASE, "none", id="gpt-5.6-explicit-none-stays-chat"),
|
||||
pytest.param("gpt-5.5", _FOUNDRY_API_BASE, "high", id="gpt-5.5-explicit-effort-stays-chat"),
|
||||
pytest.param("gpt-5.4-mini", _FOUNDRY_API_BASE, None, id="gpt-5.4-mini-unset-effort-stays-chat"),
|
||||
pytest.param("gpt-5.4-mini", _FOUNDRY_API_BASE, "low", id="gpt-5.4-mini-explicit-effort-stays-chat"),
|
||||
pytest.param("gpt-6-astra", "https://myproject.models.ai.azure.com", None, id="serverless-host-stays-chat"),
|
||||
pytest.param("Mistral-large-2411", _FOUNDRY_API_BASE, None, id="non-gpt-5-model-stays-chat"),
|
||||
pytest.param("claude-opus-4-1", _FOUNDRY_API_BASE, None, id="claude-on-foundry-stays-chat"),
|
||||
],
|
||||
)
|
||||
def test_responses_api_bridge_check_azure_ai_without_foundry_responses_route_stays_chat(
|
||||
model_name, api_base, reasoning_effort
|
||||
):
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model=model_name,
|
||||
custom_llm_provider="azure_ai",
|
||||
tools=_FOUNDRY_FUNCTION_TOOL,
|
||||
reasoning_effort=reasoning_effort,
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert model == model_name
|
||||
assert model_info.get("mode") != "responses"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_older_gpt_5_tools_without_reasoning_stays_chat():
|
||||
"""Pre-5.4 GPT-5 names keep the old boundary: tools alone never bridge."""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
|
@ -1488,6 +1579,81 @@ def test_responses_bridge_preserves_reasoning_effort_with_drop_params(
|
|||
assert request_body["reasoning"] == {"effort": "high"}
|
||||
|
||||
|
||||
_FOUNDRY_RESPONSES_FUNCTION_CALL_BODY: Final = {
|
||||
"id": "resp_foundry",
|
||||
"object": "response",
|
||||
"created_at": 1789852145,
|
||||
"status": "completed",
|
||||
"model": "gpt-6-astra",
|
||||
"output": [
|
||||
{
|
||||
"id": "fc_1",
|
||||
"type": "function_call",
|
||||
"status": "completed",
|
||||
"arguments": '{"city":"Paris"}',
|
||||
"call_id": "call_1",
|
||||
"name": "get_weather",
|
||||
}
|
||||
],
|
||||
"parallel_tool_calls": True,
|
||||
"usage": {
|
||||
"input_tokens": 53,
|
||||
"output_tokens": 18,
|
||||
"total_tokens": 71,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
},
|
||||
"error": None,
|
||||
"incomplete_details": None,
|
||||
"instructions": None,
|
||||
"metadata": {},
|
||||
"temperature": 1.0,
|
||||
"tool_choice": "auto",
|
||||
"tools": [],
|
||||
"top_p": 1.0,
|
||||
"max_output_tokens": 200,
|
||||
"previous_response_id": None,
|
||||
"reasoning": {"effort": "medium", "summary": None},
|
||||
"truncation": "disabled",
|
||||
"user": None,
|
||||
}
|
||||
|
||||
|
||||
def test_completion_bridges_azure_ai_foundry_gpt_5_4_plus_function_tools_to_responses(
|
||||
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
||||
responses_route: Final = respx_mock.post(f"{_FOUNDRY_API_BASE}/openai/v1/responses").respond(
|
||||
json=_FOUNDRY_RESPONSES_FUNCTION_CALL_BODY
|
||||
)
|
||||
|
||||
response: Final = litellm.completion(
|
||||
model="azure_ai/gpt-6-astra",
|
||||
messages=[{"role": "user", "content": "What is the weather in Paris? Use the tool."}],
|
||||
tools=[
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get weather for a city",
|
||||
"parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]},
|
||||
},
|
||||
}
|
||||
],
|
||||
max_tokens=200,
|
||||
api_base=_FOUNDRY_API_BASE,
|
||||
api_key="fake-foundry-key",
|
||||
)
|
||||
|
||||
assert [str(call.request.url) for call in respx_mock.calls] == [f"{_FOUNDRY_API_BASE}/openai/v1/responses"]
|
||||
request: Final = responses_route.calls[0].request
|
||||
request_body: Final = json.loads(request.content)
|
||||
assert request_body["tools"][0]["type"] == "function"
|
||||
assert request_body["tools"][0]["name"] == "get_weather"
|
||||
assert request.headers["api-key"] == "fake-foundry-key"
|
||||
assert response.choices[0].finish_reason == "tool_calls"
|
||||
assert response.choices[0].message.tool_calls[0].function.name == "get_weather"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, model_info, expected_model_param, expected_base_model_param",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue