diff --git a/litellm/main.py b/litellm/main.py index 1cac723c179..a1b41315b85 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -416,7 +416,8 @@ async def acompletion( logprobs: bool | None = None, top_logprobs: int | None = None, deployment_id=None, - reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] | None = None, + reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] + | None = None, verbosity: Literal["low", "medium", "high"] | None = None, safety_identifier: str | None = None, service_tier: str | None = None, @@ -4920,7 +4921,8 @@ def completion( logit_bias: dict | None = None, user: str | None = None, # openai v1.0+ new params - reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] | None = None, + reasoning_effort: Literal["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "default"] + | None = None, verbosity: Literal["low", "medium", "high"] | None = None, response_format: dict | type[BaseModel] | None = None, seed: int | None = None, diff --git a/litellm/router_utils/reasoning_effort_capability.py b/litellm/router_utils/reasoning_effort_capability.py index f59c53e9287..593b5599703 100644 --- a/litellm/router_utils/reasoning_effort_capability.py +++ b/litellm/router_utils/reasoning_effort_capability.py @@ -3,9 +3,15 @@ The model-map flags carry different polarity per level, mirroring the provider gates (gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out; anthropic/chat/transformation.py rejects only xhigh/max without an explicit flag): medium and high -are unconditional for any reasoning model, none/minimal/low are supported unless the map explicitly +are unconditional for any reasoning model, minimal/low are supported unless the map explicitly says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that polarity in one place instead of re-encoding it in every consumer. + +The none level is the one flag whose polarity is provider-dependent. OpenAI never refuses it on the +request path (azure/chat/gpt_5_transformation.py is the only caller that does, and it raises +UnsupportedParamsError unless supports_none_reasoning_effort is explicitly true), so none is opt-out +everywhere except azure, where it is opt-in. Resolving it the other way for azure would advertise a +level litellm itself rejects, which is the failure this module exists to prevent. """ from collections.abc import Mapping, Sequence @@ -14,7 +20,6 @@ from typing import Final REASONING_EFFORT_CAPABILITY_ORDER: Final = ("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra") _OPT_OUT_FLAGS: Final = ( - ("none", "supports_none_reasoning_effort"), ("minimal", "supports_minimal_reasoning_effort"), ("low", "supports_low_reasoning_effort"), ) @@ -24,6 +29,28 @@ _OPT_IN_FLAGS: Final = ( ("ultra", "supports_ultra_reasoning_effort"), ) _UNCONDITIONAL_EFFORTS: Final = frozenset(("medium", "high")) +_NONE_FLAG: Final = "supports_none_reasoning_effort" +_NONE_OPT_IN_PROVIDERS: Final = frozenset(("azure",)) + + +def _supports_none_reasoning_effort(model_info: Mapping[str, object]) -> bool: + """Opt-in on azure, whose gpt-5 config raises on reasoning_effort='none' without an explicit + true; opt-out elsewhere, where no request path refuses the level. A missing azure flag defers to + _supports_factory, the same resolver the azure gate calls, so its bare-model-name fallback + (azure/gpt-5.2 inheriting the flag from gpt-5.2) reaches both sides alike.""" + flag: Final = model_info.get(_NONE_FLAG) + if model_info.get("litellm_provider") not in _NONE_OPT_IN_PROVIDERS: + return flag is not False + if flag is not None: + return flag is True + model_key: Final = model_info.get("key") + if not isinstance(model_key, str): + return False + from litellm.utils import ( + _supports_factory, # pyright: ignore[reportPrivateUsage] # the resolver the azure gate itself calls; a public wrapper would fork the fallback + ) + + return _supports_factory(model=model_key, custom_llm_provider=None, key=_NONE_FLAG) def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None: @@ -35,7 +62,8 @@ def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tup return () opt_out: Final = frozenset(effort for effort, flag in _OPT_OUT_FLAGS if model_info.get(flag) is not False) opt_in: Final = frozenset(effort for effort, flag in _OPT_IN_FLAGS if model_info.get(flag) is True) - allowed: Final = opt_out | _UNCONDITIONAL_EFFORTS | opt_in + none_level: Final = frozenset(("none",)) if _supports_none_reasoning_effort(model_info) else frozenset() + allowed: Final = opt_out | _UNCONDITIONAL_EFFORTS | opt_in | none_level return tuple(effort for effort in REASONING_EFFORT_CAPABILITY_ORDER if effort in allowed) diff --git a/tests/test_litellm/router_utils/test_reasoning_effort_capability.py b/tests/test_litellm/router_utils/test_reasoning_effort_capability.py index ebc7952081c..1c7d5355549 100644 --- a/tests/test_litellm/router_utils/test_reasoning_effort_capability.py +++ b/tests/test_litellm/router_utils/test_reasoning_effort_capability.py @@ -1,3 +1,5 @@ +import pytest + from litellm.router_utils.reasoning_effort_capability import ( intersect_supported_reasoning_efforts, resolve_supported_reasoning_efforts, @@ -64,6 +66,48 @@ class TestResolveSupportedReasoningEfforts: assert "xhigh" not in resolved +class TestNoneLevelPolarity: + def test_none_stays_opt_out_off_azure(self): + resolved = resolve_supported_reasoning_efforts( + {"supports_reasoning": True, "litellm_provider": "openai", "key": "gpt-5-mini"} + ) + assert resolved is not None and "none" in resolved + + def test_azure_without_the_flag_does_not_advertise_none(self): + resolved = resolve_supported_reasoning_efforts( + {"supports_reasoning": True, "litellm_provider": "azure", "key": "azure/unmapped-deployment"} + ) + assert resolved == ("minimal", "low", "medium", "high") + + def test_azure_with_the_flag_advertises_none(self): + resolved = resolve_supported_reasoning_efforts( + { + "supports_reasoning": True, + "litellm_provider": "azure", + "supports_none_reasoning_effort": True, + "key": "azure/unmapped-deployment", + } + ) + assert resolved is not None and "none" in resolved + + @pytest.mark.parametrize( + "model_key", + ["azure/gpt-5", "azure/gpt-5-mini", "azure/gpt-5-nano", "azure/gpt-5.2", "azure/gpt-5.6"], + ) + def test_azure_advertisement_matches_the_azure_request_gate(self, model_key): + """AzureOpenAIGPT5Config raises UnsupportedParamsError on reasoning_effort='none' for models + it does not flag, so advertising the level there would offer routing a 400.""" + from litellm.llms.azure.chat.gpt_5_transformation import AzureOpenAIGPT5Config + from litellm.utils import _get_model_info_helper + + model_info = dict(_get_model_info_helper(model=model_key.split("/", 1)[1], custom_llm_provider="azure")) + resolved = resolve_supported_reasoning_efforts(model_info) + + assert resolved is not None + gate_accepts_none = AzureOpenAIGPT5Config._supports_reasoning_effort_level(model_key, "none") + assert ("none" in resolved) is gate_accepts_none + + class TestIntersectSupportedReasoningEfforts: def test_unknown_never_narrows(self): assert intersect_supported_reasoning_efforts(["medium", "high"], None) == ("medium", "high")