diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d4c5b476af6..f7b2c15abcf 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -4194,10 +4194,10 @@ "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, - "mode": "responses", + "mode": "chat", "output_cost_per_token": 1.4e-05, "supported_endpoints": [ - "/v1/responses" + "/v1/chat/completions" ], "supported_modalities": [ "text", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 4934f11d456..659b7c12e7d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -4206,10 +4206,10 @@ "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, - "mode": "responses", + "mode": "chat", "output_cost_per_token": 1.4e-05, "supported_endpoints": [ - "/v1/responses" + "/v1/chat/completions" ], "supported_modalities": [ "text", diff --git a/tests/test_litellm/llms/azure/chat/test_azure_gpt5_transformation.py b/tests/test_litellm/llms/azure/chat/test_azure_gpt5_transformation.py index 25f3d1364f6..60686199ac7 100644 --- a/tests/test_litellm/llms/azure/chat/test_azure_gpt5_transformation.py +++ b/tests/test_litellm/llms/azure/chat/test_azure_gpt5_transformation.py @@ -1,3 +1,5 @@ +from unittest.mock import patch + import pytest import litellm @@ -113,6 +115,38 @@ def test_azure_gpt5_codex_series_transform_request(config: AzureOpenAIGPT5Config assert request["model"] == "gpt-5-codex" +def test_responses_api_bridge_check_azure_gpt_53_codex_uses_chat(): + """Azure Foundry gpt-5.3-codex must use /openai/v1/chat/completions, not Responses API. + + Azure recommends openai/v1/chat/completions for gpt-5.3-codex; the Responses API + (openai/responses?api-version=...) returns 'API version not supported'. + Mode is driven by model_prices_and_context_window.json (azure/gpt-5.3-codex has mode "chat"). + We mock _get_model_info_helper so the test passes regardless of which cost map was + loaded (remote vs local backup). + """ + from litellm.main import responses_api_bridge_check + + # Mock so the test is independent of remote vs local model cost map + chat_mode_info = {"mode": "chat", "max_tokens": 128000, "litellm_provider": "azure"} + + with patch("litellm.main._get_model_info_helper") as mock_get_model_info: + mock_get_model_info.return_value = chat_mode_info + + model_info, model = responses_api_bridge_check( + model="gpt-5.3-codex", + custom_llm_provider="azure", + ) + assert model_info["mode"] == "chat" + assert model == "gpt-5.3-codex" + + model_info2, model2 = responses_api_bridge_check( + model="azure/gpt-5.3-codex", + custom_llm_provider="azure", + ) + assert model_info2["mode"] == "chat" + assert model2 == "azure/gpt-5.3-codex" + + # GPT-5.1 temperature handling tests for Azure def test_azure_gpt5_1_temperature_with_reasoning_effort_none(config: AzureOpenAIGPT5Config): """Test that Azure GPT-5.1 supports any temperature when reasoning_effort='none'.