mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(router): resolve the none reasoning effort as opt-in on azure
AzureOpenAIGPT5Config raises UnsupportedParamsError on reasoning_effort='none' unless the model map flags it, while OpenAI never refuses the level, so a single opt-out polarity advertised none for 61 azure deployments that reject it. Defer to _supports_factory when the azure flag is absent so the advertised list and the request gate agree on every azure gpt-5 model in the map. Also wrap the widened reasoning_effort Literal in main.py, which ruff format flagged over the line limit.
This commit is contained in:
parent
a5dd020023
commit
b0fee551fd
3 changed files with 79 additions and 5 deletions
|
|
@ -416,7 +416,8 @@ async def acompletion(
|
|||
logprobs: bool | None = None,
|
||||
top_logprobs: int | None = None,
|
||||
deployment_id=None,
|
||||
reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] | None = None,
|
||||
reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"]
|
||||
| None = None,
|
||||
verbosity: Literal["low", "medium", "high"] | None = None,
|
||||
safety_identifier: str | None = None,
|
||||
service_tier: str | None = None,
|
||||
|
|
@ -4920,7 +4921,8 @@ def completion(
|
|||
logit_bias: dict | None = None,
|
||||
user: str | None = None,
|
||||
# openai v1.0+ new params
|
||||
reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] | None = None,
|
||||
reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"]
|
||||
| None = None,
|
||||
verbosity: Literal["low", "medium", "high"] | None = None,
|
||||
response_format: dict | type[BaseModel] | None = None,
|
||||
seed: int | None = None,
|
||||
|
|
|
|||
|
|
@ -3,9 +3,15 @@
|
|||
The model-map flags carry different polarity per level, mirroring the provider gates
|
||||
(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out;
|
||||
anthropic/chat/transformation.py rejects only xhigh/max without an explicit flag): medium and high
|
||||
are unconditional for any reasoning model, none/minimal/low are supported unless the map explicitly
|
||||
are unconditional for any reasoning model, minimal/low are supported unless the map explicitly
|
||||
says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that polarity
|
||||
in one place instead of re-encoding it in every consumer.
|
||||
|
||||
The none level is the one flag whose polarity is provider-dependent. OpenAI never refuses it on the
|
||||
request path (azure/chat/gpt_5_transformation.py is the only caller that does, and it raises
|
||||
UnsupportedParamsError unless supports_none_reasoning_effort is explicitly true), so none is opt-out
|
||||
everywhere except azure, where it is opt-in. Resolving it the other way for azure would advertise a
|
||||
level litellm itself rejects, which is the failure this module exists to prevent.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
|
|
@ -14,7 +20,6 @@ from typing import Final
|
|||
REASONING_EFFORT_CAPABILITY_ORDER: Final = ("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra")
|
||||
|
||||
_OPT_OUT_FLAGS: Final = (
|
||||
("none", "supports_none_reasoning_effort"),
|
||||
("minimal", "supports_minimal_reasoning_effort"),
|
||||
("low", "supports_low_reasoning_effort"),
|
||||
)
|
||||
|
|
@ -24,6 +29,28 @@ _OPT_IN_FLAGS: Final = (
|
|||
("ultra", "supports_ultra_reasoning_effort"),
|
||||
)
|
||||
_UNCONDITIONAL_EFFORTS: Final = frozenset(("medium", "high"))
|
||||
_NONE_FLAG: Final = "supports_none_reasoning_effort"
|
||||
_NONE_OPT_IN_PROVIDERS: Final = frozenset(("azure",))
|
||||
|
||||
|
||||
def _supports_none_reasoning_effort(model_info: Mapping[str, object]) -> bool:
|
||||
"""Opt-in on azure, whose gpt-5 config raises on reasoning_effort='none' without an explicit
|
||||
true; opt-out elsewhere, where no request path refuses the level. A missing azure flag defers to
|
||||
_supports_factory, the same resolver the azure gate calls, so its bare-model-name fallback
|
||||
(azure/gpt-5.2 inheriting the flag from gpt-5.2) reaches both sides alike."""
|
||||
flag: Final = model_info.get(_NONE_FLAG)
|
||||
if model_info.get("litellm_provider") not in _NONE_OPT_IN_PROVIDERS:
|
||||
return flag is not False
|
||||
if flag is not None:
|
||||
return flag is True
|
||||
model_key: Final = model_info.get("key")
|
||||
if not isinstance(model_key, str):
|
||||
return False
|
||||
from litellm.utils import (
|
||||
_supports_factory, # pyright: ignore[reportPrivateUsage] # the resolver the azure gate itself calls; a public wrapper would fork the fallback
|
||||
)
|
||||
|
||||
return _supports_factory(model=model_key, custom_llm_provider=None, key=_NONE_FLAG)
|
||||
|
||||
|
||||
def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None:
|
||||
|
|
@ -35,7 +62,8 @@ def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tup
|
|||
return ()
|
||||
opt_out: Final = frozenset(effort for effort, flag in _OPT_OUT_FLAGS if model_info.get(flag) is not False)
|
||||
opt_in: Final = frozenset(effort for effort, flag in _OPT_IN_FLAGS if model_info.get(flag) is True)
|
||||
allowed: Final = opt_out | _UNCONDITIONAL_EFFORTS | opt_in
|
||||
none_level: Final = frozenset(("none",)) if _supports_none_reasoning_effort(model_info) else frozenset()
|
||||
allowed: Final = opt_out | _UNCONDITIONAL_EFFORTS | opt_in | none_level
|
||||
return tuple(effort for effort in REASONING_EFFORT_CAPABILITY_ORDER if effort in allowed)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
import pytest
|
||||
|
||||
from litellm.router_utils.reasoning_effort_capability import (
|
||||
intersect_supported_reasoning_efforts,
|
||||
resolve_supported_reasoning_efforts,
|
||||
|
|
@ -64,6 +66,48 @@ class TestResolveSupportedReasoningEfforts:
|
|||
assert "xhigh" not in resolved
|
||||
|
||||
|
||||
class TestNoneLevelPolarity:
|
||||
def test_none_stays_opt_out_off_azure(self):
|
||||
resolved = resolve_supported_reasoning_efforts(
|
||||
{"supports_reasoning": True, "litellm_provider": "openai", "key": "gpt-5-mini"}
|
||||
)
|
||||
assert resolved is not None and "none" in resolved
|
||||
|
||||
def test_azure_without_the_flag_does_not_advertise_none(self):
|
||||
resolved = resolve_supported_reasoning_efforts(
|
||||
{"supports_reasoning": True, "litellm_provider": "azure", "key": "azure/unmapped-deployment"}
|
||||
)
|
||||
assert resolved == ("minimal", "low", "medium", "high")
|
||||
|
||||
def test_azure_with_the_flag_advertises_none(self):
|
||||
resolved = resolve_supported_reasoning_efforts(
|
||||
{
|
||||
"supports_reasoning": True,
|
||||
"litellm_provider": "azure",
|
||||
"supports_none_reasoning_effort": True,
|
||||
"key": "azure/unmapped-deployment",
|
||||
}
|
||||
)
|
||||
assert resolved is not None and "none" in resolved
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_key",
|
||||
["azure/gpt-5", "azure/gpt-5-mini", "azure/gpt-5-nano", "azure/gpt-5.2", "azure/gpt-5.6"],
|
||||
)
|
||||
def test_azure_advertisement_matches_the_azure_request_gate(self, model_key):
|
||||
"""AzureOpenAIGPT5Config raises UnsupportedParamsError on reasoning_effort='none' for models
|
||||
it does not flag, so advertising the level there would offer routing a 400."""
|
||||
from litellm.llms.azure.chat.gpt_5_transformation import AzureOpenAIGPT5Config
|
||||
from litellm.utils import _get_model_info_helper
|
||||
|
||||
model_info = dict(_get_model_info_helper(model=model_key.split("/", 1)[1], custom_llm_provider="azure"))
|
||||
resolved = resolve_supported_reasoning_efforts(model_info)
|
||||
|
||||
assert resolved is not None
|
||||
gate_accepts_none = AzureOpenAIGPT5Config._supports_reasoning_effort_level(model_key, "none")
|
||||
assert ("none" in resolved) is gate_accepts_none
|
||||
|
||||
|
||||
class TestIntersectSupportedReasoningEfforts:
|
||||
def test_unknown_never_narrows(self):
|
||||
assert intersect_supported_reasoning_efforts(["medium", "high"], None) == ("medium", "high")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue