fix(azure): rename max_tokens to max_completion_tokens for gpt-5-chat deployments (#36857)

Azure rejects the legacy `max_tokens` key for the whole gpt-5 name family, but
`AzureOpenAIGPT5Config.is_model_gpt_5_model` deliberately excludes `gpt-5-chat*`
so those deployments fall through to `AzureOpenAIConfig`, which sends `max_tokens`
verbatim and gets a 400 back on every request that carries it, `/health` probes
included.

One predicate was answering two independent questions. Split it: the new
`AzureOpenAIConfig.requires_max_completion_tokens` covers the whole gpt-5 name
family and drives only the rename, while `is_model_gpt_5_model` keeps keying
reasoning_effort, the temperature clamp and the dropped penalties off the
reasoning question, so #13781 stays fixed.
This commit is contained in:
Yassin Kortam 2026-08-17 11:27:55 -07:00 • committed by GitHub
parent ee08b63657
commit 9b7ed77fcc
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 89 additions and 1 deletions

View file

@ -112,6 +112,16 @@ class AzureOpenAIConfig(BaseConfig):
"store",
]
@classmethod
def requires_max_completion_tokens(cls, model: str) -> bool:
"""Whether Azure rejects the legacy ``max_tokens`` key for this deployment.
Deliberately wider than ``AzureOpenAIGPT5Config.is_model_gpt_5_model``: the whole gpt-5
name family needs the rename, including the ``gpt-5-chat*`` models that are excluded from
the reasoning path by https://github.com/BerriAI/litellm/issues/13781.
"""
return "gpt-5" in model or "gpt5_series" in model
def _is_response_format_supported_model(self, model: str) -> bool:
"""
Determines if the model supports response_format.
@ -160,6 +170,7 @@ class AzureOpenAIConfig(BaseConfig):
api_version: str = "",
) -> dict:
supported_openai_params: Final = self.get_supported_openai_params(model)
renames_max_tokens: Final = self.requires_max_completion_tokens(model)
api_version_times: Final = api_version.split("-")
if len(api_version_times) >= 3:
@ -172,7 +183,9 @@ class AzureOpenAIConfig(BaseConfig):
api_version_day = None
for param, value in non_default_params.items():
if param == "tool_choice":
if param == "max_tokens" and renames_max_tokens:
optional_params.setdefault("max_completion_tokens", value)
elif param == "tool_choice":
"""
This parameter requires API version 2023-12-01-preview or later

View file

@ -1,12 +1,21 @@
import os
import sys
from typing import Final
import pytest
from pydantic import TypeAdapter
sys.path.insert(
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
)
import litellm
from litellm.litellm_core_utils.prompt_templates.common_utils import TOOL_RESULT_IMAGE_BOUNDARY
from litellm.llms.azure.chat.gpt_transformation import AzureOpenAIConfig
from litellm.utils import get_optional_params
_MAPPED_PARAMS: Final = TypeAdapter(dict[str, object])
_SUPPORTED_PARAMS: Final = TypeAdapter(list[str])
class TestAzureOpenAIConfig:
@ -91,3 +100,69 @@ def test_transform_request_hoists_tool_message_image():
{"type": "text", "text": TOOL_RESULT_IMAGE_BOUNDARY},
{"type": "image_url", "image_url": {"url": data_uri}},
]
@pytest.mark.parametrize(
"model, emitted_key, absent_key",
[
("gpt-5-chat", "max_completion_tokens", "max_tokens"),
("gpt-5-chat-latest", "max_completion_tokens", "max_tokens"),
("gpt-5-chat-2025-08-07", "max_completion_tokens", "max_tokens"),
("gpt-5", "max_completion_tokens", "max_tokens"),
("o3-mini", "max_completion_tokens", "max_tokens"),
("gpt-4o", "max_tokens", "max_completion_tokens"),
],
)
def test_azure_max_tokens_rename_covers_gpt_5_chat_family(model: str, emitted_key: str, absent_key: str) -> None:
"""Azure rejects `max_tokens` for the whole gpt-5 name family, gpt-5-chat* included."""
mapped: Final = _MAPPED_PARAMS.validate_python(
get_optional_params(model=model, custom_llm_provider="azure", max_tokens=5)
)
assert mapped[emitted_key] == 5
assert absent_key not in mapped
@pytest.mark.parametrize("model", ["gpt-5-chat", "gpt-5-chat-latest"])
def test_azure_gpt_5_chat_stays_off_the_reasoning_path(model: str) -> None:
"""https://github.com/BerriAI/litellm/issues/13781: gpt-5-chat* is a regular chat model."""
mapped: Final = _MAPPED_PARAMS.validate_python(
get_optional_params(
model=model,
custom_llm_provider="azure",
max_tokens=5,
temperature=0.3,
presence_penalty=0.1,
frequency_penalty=0.2,
stop=["stop"],
logit_bias={"1": 1},
)
)
supported: Final = _SUPPORTED_PARAMS.validate_python(
litellm.get_supported_openai_params(model=model, custom_llm_provider="azure")
)
assert mapped["temperature"] == 0.3
assert mapped["presence_penalty"] == 0.1
assert mapped["frequency_penalty"] == 0.2
assert mapped["stop"] == ["stop"]
assert mapped["logit_bias"] == {"1": 1}
assert "reasoning_effort" not in mapped
assert "reasoning_effort" not in supported
def test_azure_gpt_5_takes_the_reasoning_path() -> None:
"""Positive control for the predicate split: gpt-5 still drops chat-only params."""
mapped: Final = _MAPPED_PARAMS.validate_python(
get_optional_params(
model="gpt-5",
custom_llm_provider="azure",
presence_penalty=0.1,
logit_bias={"1": 1},
drop_params=True,
)
)
supported: Final = _SUPPORTED_PARAMS.validate_python(
litellm.get_supported_openai_params(model="gpt-5", custom_llm_provider="azure")
)
assert "presence_penalty" not in mapped
assert "logit_bias" not in mapped
assert "reasoning_effort" in supported