fix(azure_ai): strip the azure_ai/ prefix when a Responses call is remapped to azure

A catalog OpenAI name on an .openai.azure.com host (or with AZURE_AI_API_BASE set to one) is remapped from azure_ai to azure before the Responses request is built, and the azure_ai/ prefix stayed in the wire model, so Azure answered DeploymentNotFound. The Azure Responses config now strips azure_ai/ next to responses/ and o_series/.
This commit is contained in:
mateo-berri 2026-09-16 16:49:29 -07:00
parent ebae692a0d
commit ada0a1ad3a
3 changed files with 30 additions and 6 deletions

View file

@ -49,12 +49,7 @@ class AzureOpenAIResponsesAPIConfig(OpenAIResponsesAPIConfig):
return BaseAzureLLM._base_validate_azure_environment(headers=headers, litellm_params=litellm_params)
def get_stripped_model_name(self, model: str) -> str:
# if "responses/" is in the model name, remove it
if "responses/" in model:
model = model.replace("responses/", "")
if "o_series" in model:
model = model.replace("o_series/", "")
return model
return model.replace("responses/", "").replace("o_series/", "").replace("azure_ai/", "")
def _handle_reasoning_item(self, item: dict[str, Any]) -> dict[str, Any]:
"""

View file

@ -677,3 +677,14 @@ def test_azure_responses_gpt6_astra_rejects_temperature_while_reasoning(local_mo
model="gpt-6-astra",
drop_params=False,
)
def test_azure_responses_sends_the_deployment_name_when_azure_ai_prefix_survives_provider_remap():
request = AzureOpenAIResponsesAPIConfig().transform_responses_api_request(
model="azure_ai/gpt-5.4-nano",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert request["model"] == "gpt-5.4-nano"

View file

@ -253,6 +253,24 @@ async def test_aresponses_sends_reasoning_and_tools_to_native_endpoint(model, ap
_assert_native_responses_request(route, expected_url, expected_model)
@pytest.mark.asyncio
@respx.mock
async def test_aresponses_catalog_name_remapped_to_azure_sends_bare_deployment_name(monkeypatch):
monkeypatch.setenv("AZURE_AI_API_BASE", "https://res.openai.azure.com")
route = respx.post(url__regex=r".*/openai/v1/responses(\?.*)?$").mock(
return_value=httpx.Response(200, json=_responses_payload("gpt-5.4-nano"))
)
await litellm.aresponses(
model="azure_ai/gpt-5.4-nano",
input="What is the weather in SF?",
api_base="https://res.openai.azure.com",
api_key="fake-key",
)
assert json.loads(route.calls.last.request.content)["model"] == "gpt-5.4-nano"
@pytest.mark.asyncio
@respx.mock
@pytest.mark.parametrize("model,api_base,expected_url,expected_model", NATIVE_RESPONSES_CASES)