fix(azure): route gpt-5.3-codex to chat completions instead of Responses API

This commit is contained in:
Varad Khonde 2026-03-03 17:43:17 +05:30
parent 67f90254ed
commit 70278c3b1c
3 changed files with 38 additions and 4 deletions

View file

@ -4194,10 +4194,10 @@
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"supported_endpoints": [
"/v1/responses"
"/v1/chat/completions"
],
"supported_modalities": [
"text",

View file

@ -4206,10 +4206,10 @@
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "responses",
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"supported_endpoints": [
"/v1/responses"
"/v1/chat/completions"
],
"supported_modalities": [
"text",

View file

@ -1,3 +1,5 @@
from unittest.mock import patch
import pytest
import litellm
@ -113,6 +115,38 @@ def test_azure_gpt5_codex_series_transform_request(config: AzureOpenAIGPT5Config
assert request["model"] == "gpt-5-codex"
def test_responses_api_bridge_check_azure_gpt_53_codex_uses_chat():
"""Azure Foundry gpt-5.3-codex must use /openai/v1/chat/completions, not Responses API.
Azure recommends openai/v1/chat/completions for gpt-5.3-codex; the Responses API
(openai/responses?api-version=...) returns 'API version not supported'.
Mode is driven by model_prices_and_context_window.json (azure/gpt-5.3-codex has mode "chat").
We mock _get_model_info_helper so the test passes regardless of which cost map was
loaded (remote vs local backup).
"""
from litellm.main import responses_api_bridge_check
# Mock so the test is independent of remote vs local model cost map
chat_mode_info = {"mode": "chat", "max_tokens": 128000, "litellm_provider": "azure"}
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
mock_get_model_info.return_value = chat_mode_info
model_info, model = responses_api_bridge_check(
model="gpt-5.3-codex",
custom_llm_provider="azure",
)
assert model_info["mode"] == "chat"
assert model == "gpt-5.3-codex"
model_info2, model2 = responses_api_bridge_check(
model="azure/gpt-5.3-codex",
custom_llm_provider="azure",
)
assert model_info2["mode"] == "chat"
assert model2 == "azure/gpt-5.3-codex"
# GPT-5.1 temperature handling tests for Azure
def test_azure_gpt5_1_temperature_with_reasoning_effort_none(config: AzureOpenAIGPT5Config):
"""Test that Azure GPT-5.1 supports any temperature when reasoning_effort='none'.