mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(azure): rename max_tokens to max_completion_tokens for gpt-5-chat deployments (#36857)
Azure rejects the legacy `max_tokens` key for the whole gpt-5 name family, but `AzureOpenAIGPT5Config.is_model_gpt_5_model` deliberately excludes `gpt-5-chat*` so those deployments fall through to `AzureOpenAIConfig`, which sends `max_tokens` verbatim and gets a 400 back on every request that carries it, `/health` probes included. One predicate was answering two independent questions. Split it: the new `AzureOpenAIConfig.requires_max_completion_tokens` covers the whole gpt-5 name family and drives only the rename, while `is_model_gpt_5_model` keeps keying reasoning_effort, the temperature clamp and the dropped penalties off the reasoning question, so #13781 stays fixed.
This commit is contained in:
parent
ee08b63657
commit
9b7ed77fcc
2 changed files with 89 additions and 1 deletions
|
|
@ -112,6 +112,16 @@ class AzureOpenAIConfig(BaseConfig):
|
|||
"store",
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def requires_max_completion_tokens(cls, model: str) -> bool:
|
||||
"""Whether Azure rejects the legacy ``max_tokens`` key for this deployment.
|
||||
|
||||
Deliberately wider than ``AzureOpenAIGPT5Config.is_model_gpt_5_model``: the whole gpt-5
|
||||
name family needs the rename, including the ``gpt-5-chat*`` models that are excluded from
|
||||
the reasoning path by https://github.com/BerriAI/litellm/issues/13781.
|
||||
"""
|
||||
return "gpt-5" in model or "gpt5_series" in model
|
||||
|
||||
def _is_response_format_supported_model(self, model: str) -> bool:
|
||||
"""
|
||||
Determines if the model supports response_format.
|
||||
|
|
@ -160,6 +170,7 @@ class AzureOpenAIConfig(BaseConfig):
|
|||
api_version: str = "",
|
||||
) -> dict:
|
||||
supported_openai_params: Final = self.get_supported_openai_params(model)
|
||||
renames_max_tokens: Final = self.requires_max_completion_tokens(model)
|
||||
api_version_times: Final = api_version.split("-")
|
||||
|
||||
if len(api_version_times) >= 3:
|
||||
|
|
@ -172,7 +183,9 @@ class AzureOpenAIConfig(BaseConfig):
|
|||
api_version_day = None
|
||||
|
||||
for param, value in non_default_params.items():
|
||||
if param == "tool_choice":
|
||||
if param == "max_tokens" and renames_max_tokens:
|
||||
optional_params.setdefault("max_completion_tokens", value)
|
||||
elif param == "tool_choice":
|
||||
"""
|
||||
This parameter requires API version 2023-12-01-preview or later
|
||||
|
||||
|
|
|
|||
|
|
@ -1,12 +1,21 @@
|
|||
import os
|
||||
import sys
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
|
||||
)
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import TOOL_RESULT_IMAGE_BOUNDARY
|
||||
from litellm.llms.azure.chat.gpt_transformation import AzureOpenAIConfig
|
||||
from litellm.utils import get_optional_params
|
||||
|
||||
_MAPPED_PARAMS: Final = TypeAdapter(dict[str, object])
|
||||
_SUPPORTED_PARAMS: Final = TypeAdapter(list[str])
|
||||
|
||||
|
||||
class TestAzureOpenAIConfig:
|
||||
|
|
@ -91,3 +100,69 @@ def test_transform_request_hoists_tool_message_image():
|
|||
{"type": "text", "text": TOOL_RESULT_IMAGE_BOUNDARY},
|
||||
{"type": "image_url", "image_url": {"url": data_uri}},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, emitted_key, absent_key",
|
||||
[
|
||||
("gpt-5-chat", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-5-chat-latest", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-5-chat-2025-08-07", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-5", "max_completion_tokens", "max_tokens"),
|
||||
("o3-mini", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-4o", "max_tokens", "max_completion_tokens"),
|
||||
],
|
||||
)
|
||||
def test_azure_max_tokens_rename_covers_gpt_5_chat_family(model: str, emitted_key: str, absent_key: str) -> None:
|
||||
"""Azure rejects `max_tokens` for the whole gpt-5 name family, gpt-5-chat* included."""
|
||||
mapped: Final = _MAPPED_PARAMS.validate_python(
|
||||
get_optional_params(model=model, custom_llm_provider="azure", max_tokens=5)
|
||||
)
|
||||
assert mapped[emitted_key] == 5
|
||||
assert absent_key not in mapped
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5-chat", "gpt-5-chat-latest"])
|
||||
def test_azure_gpt_5_chat_stays_off_the_reasoning_path(model: str) -> None:
|
||||
"""https://github.com/BerriAI/litellm/issues/13781: gpt-5-chat* is a regular chat model."""
|
||||
mapped: Final = _MAPPED_PARAMS.validate_python(
|
||||
get_optional_params(
|
||||
model=model,
|
||||
custom_llm_provider="azure",
|
||||
max_tokens=5,
|
||||
temperature=0.3,
|
||||
presence_penalty=0.1,
|
||||
frequency_penalty=0.2,
|
||||
stop=["stop"],
|
||||
logit_bias={"1": 1},
|
||||
)
|
||||
)
|
||||
supported: Final = _SUPPORTED_PARAMS.validate_python(
|
||||
litellm.get_supported_openai_params(model=model, custom_llm_provider="azure")
|
||||
)
|
||||
assert mapped["temperature"] == 0.3
|
||||
assert mapped["presence_penalty"] == 0.1
|
||||
assert mapped["frequency_penalty"] == 0.2
|
||||
assert mapped["stop"] == ["stop"]
|
||||
assert mapped["logit_bias"] == {"1": 1}
|
||||
assert "reasoning_effort" not in mapped
|
||||
assert "reasoning_effort" not in supported
|
||||
|
||||
|
||||
def test_azure_gpt_5_takes_the_reasoning_path() -> None:
|
||||
"""Positive control for the predicate split: gpt-5 still drops chat-only params."""
|
||||
mapped: Final = _MAPPED_PARAMS.validate_python(
|
||||
get_optional_params(
|
||||
model="gpt-5",
|
||||
custom_llm_provider="azure",
|
||||
presence_penalty=0.1,
|
||||
logit_bias={"1": 1},
|
||||
drop_params=True,
|
||||
)
|
||||
)
|
||||
supported: Final = _SUPPORTED_PARAMS.validate_python(
|
||||
litellm.get_supported_openai_params(model="gpt-5", custom_llm_provider="azure")
|
||||
)
|
||||
assert "presence_penalty" not in mapped
|
||||
assert "logit_bias" not in mapped
|
||||
assert "reasoning_effort" in supported
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue