mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(azure): route gpt-5.3-codex to chat completions instead of Responses API
This commit is contained in:
parent
67f90254ed
commit
70278c3b1c
3 changed files with 38 additions and 4 deletions
|
|
@ -4194,10 +4194,10 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "responses",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
|
|||
|
|
@ -4206,10 +4206,10 @@
|
|||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "responses",
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.4e-05,
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
|
|
@ -113,6 +115,38 @@ def test_azure_gpt5_codex_series_transform_request(config: AzureOpenAIGPT5Config
|
|||
assert request["model"] == "gpt-5-codex"
|
||||
|
||||
|
||||
def test_responses_api_bridge_check_azure_gpt_53_codex_uses_chat():
|
||||
"""Azure Foundry gpt-5.3-codex must use /openai/v1/chat/completions, not Responses API.
|
||||
|
||||
Azure recommends openai/v1/chat/completions for gpt-5.3-codex; the Responses API
|
||||
(openai/responses?api-version=...) returns 'API version not supported'.
|
||||
Mode is driven by model_prices_and_context_window.json (azure/gpt-5.3-codex has mode "chat").
|
||||
We mock _get_model_info_helper so the test passes regardless of which cost map was
|
||||
loaded (remote vs local backup).
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
# Mock so the test is independent of remote vs local model cost map
|
||||
chat_mode_info = {"mode": "chat", "max_tokens": 128000, "litellm_provider": "azure"}
|
||||
|
||||
with patch("litellm.main._get_model_info_helper") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = chat_mode_info
|
||||
|
||||
model_info, model = responses_api_bridge_check(
|
||||
model="gpt-5.3-codex",
|
||||
custom_llm_provider="azure",
|
||||
)
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model == "gpt-5.3-codex"
|
||||
|
||||
model_info2, model2 = responses_api_bridge_check(
|
||||
model="azure/gpt-5.3-codex",
|
||||
custom_llm_provider="azure",
|
||||
)
|
||||
assert model_info2["mode"] == "chat"
|
||||
assert model2 == "azure/gpt-5.3-codex"
|
||||
|
||||
|
||||
# GPT-5.1 temperature handling tests for Azure
|
||||
def test_azure_gpt5_1_temperature_with_reasoning_effort_none(config: AzureOpenAIGPT5Config):
|
||||
"""Test that Azure GPT-5.1 supports any temperature when reasoning_effort='none'.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue