fix(azure): responses none-effort temperature gate reads the azure/ cost-map entry

This commit is contained in:
mateo-berri 2026-09-04 17:45:06 -07:00
parent 51514b9123
commit c02f198a4a
2 changed files with 48 additions and 4 deletions

View file

@ -6,6 +6,7 @@ from openai.types.responses import ResponseReasoningItem
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.url_utils import encode_url_path_segment
from litellm.llms.azure.chat.gpt_5_transformation import AzureOpenAIGPT5Config
from litellm.llms.azure.common_utils import BaseAzureLLM
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.types.llms.openai import *
@ -29,6 +30,14 @@ class AzureOpenAIResponsesAPIConfig(OpenAIResponsesAPIConfig):
def custom_llm_provider(self) -> LlmProviders:
return LlmProviders.AZURE
@staticmethod
def _supports_reasoning_effort_none(model: str) -> bool:
return AzureOpenAIGPT5Config._supports_reasoning_effort_level(model, "none")
@staticmethod
def _effort_resolves_to_none(model: str, effort: str | None) -> bool:
return AzureOpenAIGPT5Config.effort_resolves_to_none(model, effort)
def get_supported_openai_params(self, model: str) -> list:
"""
Azure Responses API does not support context_management (compaction).

View file

@ -1,11 +1,10 @@
from copy import deepcopy
from unittest.mock import patch
from unittest.mock import MagicMock, patch
import pytest
from unittest.mock import MagicMock
import litellm
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
from litellm.llms.azure.responses.o_series_transformation import (
AzureOpenAIOSeriesResponsesAPIConfig,
)
@ -613,3 +612,39 @@ class TestAzureResponsesAPIConfig:
assert result["tools"][0] is tool
assert "anyOf" in result["tools"][0]["parameters"]
@pytest.fixture()
def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
"""Pin the bundled cost map: the published map lags a key added in this repo."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", get_model_cost_map(url=litellm.model_cost_map_url))
litellm.add_known_models(model_cost_map=litellm.model_cost)
def test_azure_responses_gpt6_astra_reasoning_effort_none_unlocks_temperature(local_model_cost_map: None):
"""Foundry's gpt-6-astra accepts reasoning.effort='none' with a non-default temperature
while OpenAI's gpt-6-astra does not, so the gate must read the azure/ cost-map entry
for the bare deployment name rather than OpenAI's."""
params = AzureOpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params=ResponsesAPIOptionalRequestParams(
temperature=0.2,
reasoning={"effort": "none"},
),
model="gpt-6-astra",
drop_params=False,
)
assert params["temperature"] == 0.2
assert params["reasoning"] == {"effort": "none"}
def test_azure_responses_gpt6_astra_rejects_temperature_while_reasoning(local_model_cost_map: None):
with pytest.raises(litellm.UnsupportedParamsError):
AzureOpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params=ResponsesAPIOptionalRequestParams(
temperature=0.2,
reasoning={"effort": "low"},
),
model="gpt-6-astra",
drop_params=False,
)