fix(azure_ai): route Responses API to native /openai/v1/responses for Foundry Models

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
shivam 2026-07-18 21:55:44 +00:00
parent e238e89537
commit 518c4b07a7
9 changed files with 279 additions and 6 deletions

View file

@ -1742,6 +1742,9 @@ if TYPE_CHECKING:
from .llms.azure.responses.o_series_transformation import (
AzureOpenAIOSeriesResponsesAPIConfig as AzureOpenAIOSeriesResponsesAPIConfig,
)
from .llms.azure_ai.responses.transformation import (
AzureAIResponsesAPIConfig as AzureAIResponsesAPIConfig,
)
from .llms.xai.responses.transformation import (
XAIResponsesAPIConfig as XAIResponsesAPIConfig,
)

View file

@ -231,6 +231,7 @@ LLM_CONFIG_NAMES = (
"OpenAIResponsesAPIConfig",
"AzureOpenAIResponsesAPIConfig",
"AzureOpenAIOSeriesResponsesAPIConfig",
"AzureAIResponsesAPIConfig",
"XAIResponsesAPIConfig",
"LiteLLMProxyResponsesAPIConfig",
"HostedVLLMResponsesAPIConfig",
@ -935,6 +936,10 @@ _LLM_CONFIGS_IMPORT_MAP = {
".llms.azure.responses.o_series_transformation",
"AzureOpenAIOSeriesResponsesAPIConfig",
),
"AzureAIResponsesAPIConfig": (
".llms.azure_ai.responses.transformation",
"AzureAIResponsesAPIConfig",
),
"XAIResponsesAPIConfig": (
".llms.xai.responses.transformation",
"XAIResponsesAPIConfig",

View file

@ -1,7 +1,6 @@
import enum
import re
from typing import Any, List, Optional, Tuple, cast
from urllib.parse import urlparse
import httpx
from httpx import Response
@ -12,6 +11,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
_audio_or_image_in_message_content,
convert_content_list_to_str,
)
from litellm.llms.azure_ai.common_utils import azure_ai_use_api_key_header
from litellm.llms.azure.common_utils import BaseAzureLLM
from litellm.llms.base_llm.chat.transformation import LiteLLMLoggingObj
from litellm.llms.openai.common_utils import drop_params_from_unprocessable_entity_error
@ -85,11 +85,7 @@ class AzureAIStudioConfig(OpenAIConfig):
"""
Returns True if the request should use `api-key` header for authentication.
"""
parsed_url = urlparse(api_base)
host = parsed_url.hostname
if host and (host.endswith(".services.ai.azure.com") or host.endswith(".openai.azure.com")):
return True
return False
return azure_ai_use_api_key_header(api_base)
def get_complete_url(
self,

View file

@ -1,4 +1,5 @@
from typing import List, Literal, Optional
from urllib.parse import urlparse
import litellm
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
@ -6,6 +7,31 @@ from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllMessageValues
def azure_ai_use_api_key_header(api_base: str) -> bool:
"""Whether Azure AI auth should use the `api-key` header instead of a Bearer token.
Foundry and Azure OpenAI hosts authenticate key-based requests with the
`api-key` header; serverless/other endpoints expect `Authorization: Bearer`.
"""
host = urlparse(api_base).hostname
return bool(host and (host.endswith(".services.ai.azure.com") or host.endswith(".openai.azure.com")))
def azure_ai_supports_native_responses(model: str | None) -> bool:
"""Whether an Azure AI model should use the native Responses API rather than the chat bridge.
Foundry Models expose an OpenAI-compatible Responses endpoint at
`<endpoint>/openai/v1/responses`. Claude deployments speak the Anthropic
Messages API and the model-router/agents routes have their own surfaces, so
those keep the chat-completions bridge.
"""
if not model:
return False
if "claude" in model.lower():
return False
return AzureFoundryModelInfo.get_azure_ai_route(model) == "default"
class AzureFoundryModelInfo(BaseLLMModelInfo):
"""Model info for Azure AI / Azure Foundry models."""

View file

@ -0,0 +1,62 @@
import httpx
from litellm.llms.azure.common_utils import BaseAzureLLM
from litellm.llms.azure.responses.transformation import AzureOpenAIResponsesAPIConfig
from litellm.llms.azure_ai.common_utils import (
AzureFoundryModelInfo,
azure_ai_use_api_key_header,
)
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
from litellm.utils import _add_path_to_api_base
class AzureAIResponsesAPIConfig(AzureOpenAIResponsesAPIConfig):
"""Native Responses API config for Azure AI Foundry Models.
Foundry Models such as the GPT-5 family expose an OpenAI-compatible Responses
endpoint at `<endpoint>/openai/v1/responses`. Routing here (instead of the
chat-completions bridge) keeps `reasoning_effort` alongside function tools,
which Azure rejects on `/chat/completions`.
"""
@property
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.AZURE_AI
def validate_environment(self, headers: dict, model: str, litellm_params: GenericLiteLLMParams | None) -> dict:
litellm_params = litellm_params or GenericLiteLLMParams()
api_key = AzureFoundryModelInfo.get_api_key(litellm_params.api_key)
api_base = AzureFoundryModelInfo.get_api_base(litellm_params.api_base)
if api_key:
if api_base and azure_ai_use_api_key_header(api_base):
headers["api-key"] = api_key
else:
headers["Authorization"] = f"Bearer {api_key}"
else:
headers = BaseAzureLLM._base_validate_azure_environment(headers=headers, litellm_params=litellm_params)
headers.setdefault("Content-Type", "application/json")
return headers
def get_complete_url(
self,
api_base: str | None,
litellm_params: dict,
) -> str:
api_base = AzureFoundryModelInfo.get_api_base(api_base)
if api_base is None:
raise ValueError(
"api_base is required for Azure AI Foundry Responses API. "
"Set the api_base parameter or the AZURE_AI_API_BASE environment variable."
)
original_url = httpx.URL(api_base)
query_params = dict(original_url.params)
api_version = litellm_params.get("api_version")
if "api-version" not in query_params and isinstance(api_version, str):
query_params["api-version"] = api_version
new_url = _add_path_to_api_base(api_base=api_base, ending_path="/openai/v1/responses")
return str(httpx.URL(new_url).copy_with(params=query_params))

View file

@ -8221,6 +8221,14 @@ class ProviderConfigManager:
return litellm.AzureOpenAIOSeriesResponsesAPIConfig()
else:
return litellm.AzureOpenAIResponsesAPIConfig()
elif litellm.LlmProviders.AZURE_AI == provider:
from litellm.llms.azure_ai.common_utils import (
azure_ai_supports_native_responses,
)
if azure_ai_supports_native_responses(model):
return litellm.AzureAIResponsesAPIConfig()
return None
elif litellm.LlmProviders.XAI == provider:
return litellm.XAIResponsesAPIConfig()
elif litellm.LlmProviders.GITHUB_COPILOT == provider:

View file

@ -0,0 +1,173 @@
"""
Regression tests for native Azure AI Foundry Responses API routing (LIT-4427).
Before the fix, `azure_ai` had no native Responses config, so `litellm.responses()`
fell back to the chat-completions bridge and sent `reasoning_effort` + function tools
to `/chat/completions`, which Azure rejects for GPT-5 models. These tests assert the
request now goes to the native `/openai/v1/responses` endpoint in Responses shape.
"""
import json
from unittest.mock import AsyncMock, patch
import httpx
import pytest
import litellm
from litellm.llms.azure_ai.responses.transformation import AzureAIResponsesAPIConfig
from litellm.types.router import GenericLiteLLMParams
from litellm.utils import ProviderConfigManager
class MockResponse:
def __init__(self, json_data, status_code=200):
self._json_data = json_data
self.status_code = status_code
self.text = json.dumps(json_data)
self.headers = httpx.Headers({})
def json(self):
return self._json_data
def _minimal_responses_payload(model: str) -> dict:
return {
"id": "resp_123",
"object": "response",
"created_at": 1741369938,
"status": "completed",
"model": model,
"output": [],
"parallel_tool_calls": False,
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2},
"error": None,
"tool_choice": "auto",
"tools": [],
"metadata": None,
"temperature": None,
"top_p": None,
"max_output_tokens": None,
"previous_response_id": None,
"reasoning": None,
"truncation": None,
"instructions": None,
"incomplete_details": None,
"user": None,
}
@pytest.mark.parametrize(
"model",
["gpt-5.6-luna-20260710154139", "gpt-5.5-20260504143601", "DeepSeek-R1-0528"],
)
def test_azure_ai_resolves_native_responses_config(model):
config = ProviderConfigManager.get_provider_responses_api_config(provider="azure_ai", model=model)
assert isinstance(config, AzureAIResponsesAPIConfig)
@pytest.mark.parametrize("model", ["claude-3-5-sonnet", "model_router/gpt-5", "agents/my-agent"])
def test_azure_ai_non_responses_models_keep_bridge(model):
"""Claude / model-router / agents routes have their own surfaces, so they must
keep returning None (chat-completions bridge)."""
config = ProviderConfigManager.get_provider_responses_api_config(provider="azure_ai", model=model)
assert config is None
@pytest.mark.parametrize(
"api_base,expected",
[
(
"https://res.services.ai.azure.com/api/projects/proj",
"https://res.services.ai.azure.com/api/projects/proj/openai/v1/responses",
),
(
"https://res.services.ai.azure.com/api/projects/proj/",
"https://res.services.ai.azure.com/api/projects/proj/openai/v1/responses",
),
(
"https://res.services.ai.azure.com",
"https://res.services.ai.azure.com/openai/v1/responses",
),
(
"https://res.openai.azure.com",
"https://res.openai.azure.com/openai/v1/responses",
),
(
"https://res.services.ai.azure.com/api/projects/proj/openai/v1/responses",
"https://res.services.ai.azure.com/api/projects/proj/openai/v1/responses",
),
],
)
def test_get_complete_url(api_base, expected):
config = AzureAIResponsesAPIConfig()
assert config.get_complete_url(api_base=api_base, litellm_params={}) == expected
def test_validate_environment_api_key_header_for_foundry_host():
config = AzureAIResponsesAPIConfig()
headers = config.validate_environment(
headers={},
model="gpt-5.6-luna",
litellm_params=GenericLiteLLMParams(
api_key="secret", api_base="https://res.services.ai.azure.com/api/projects/proj"
),
)
assert headers["api-key"] == "secret"
assert "Authorization" not in headers
def test_validate_environment_bearer_for_serverless_host():
config = AzureAIResponsesAPIConfig()
headers = config.validate_environment(
headers={},
model="gpt-5.6-luna",
litellm_params=GenericLiteLLMParams(
api_key="secret", api_base="https://endpoint.eastus.models.ai.azure.com"
),
)
assert headers["Authorization"] == "Bearer secret"
assert "api-key" not in headers
@pytest.mark.asyncio
async def test_aresponses_routes_to_native_endpoint_with_reasoning_and_tools():
"""Core LIT-4427 regression: reasoning_effort + function tools must be sent to the
native /openai/v1/responses endpoint in Responses shape, not bridged to /chat/completions."""
tools = [
{
"type": "function",
"name": "get_weather",
"description": "Get weather",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
}
]
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
new_callable=AsyncMock,
) as mock_post:
mock_post.return_value = MockResponse(_minimal_responses_payload("gpt-5.6-luna"), 200)
await litellm.aresponses(
model="azure_ai/gpt-5.6-luna-20260710154139",
input="What is the weather in SF?",
reasoning_effort="high",
tools=tools,
api_base="https://res.services.ai.azure.com/api/projects/proj",
api_key="fake-key",
)
mock_post.assert_called_once()
url = str(mock_post.call_args.kwargs["url"])
body = mock_post.call_args.kwargs["json"]
assert url == "https://res.services.ai.azure.com/api/projects/proj/openai/v1/responses"
assert "/chat/completions" not in url
assert "input" in body
assert "messages" not in body
assert body["reasoning"] == {"effort": "high"}
assert body["tools"] == tools