diff --git a/litellm/llms/openai/chat/gpt_5_transformation.py b/litellm/llms/openai/chat/gpt_5_transformation.py index c55640f6a1f..3a79e6570e9 100644 --- a/litellm/llms/openai/chat/gpt_5_transformation.py +++ b/litellm/llms/openai/chat/gpt_5_transformation.py @@ -222,8 +222,8 @@ class OpenAIGPT5Config(OpenAIGPTConfig): if "reasoning_effort" in optional_params: optional_params["reasoning_effort"] = normalized - if effective_effort in ("xhigh", "max", "ultra"): - # xhigh/max/ultra are opt-in capabilities: only allow if the model explicitly supports them. + if effective_effort == "xhigh": + # xhigh is an opt-in capability: only allow if model explicitly supports it. if not self._supports_reasoning_effort_level(model, effective_effort): if litellm.drop_params or drop_params: non_default_params.pop("reasoning_effort", None) diff --git a/litellm/router_utils/reasoning_effort_capability.py b/litellm/router_utils/reasoning_effort_capability.py index 593b5599703..7c8b5523d7c 100644 --- a/litellm/router_utils/reasoning_effort_capability.py +++ b/litellm/router_utils/reasoning_effort_capability.py @@ -54,8 +54,12 @@ def _supports_none_reasoning_effort(model_info: Mapping[str, object]) -> bool: def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None: - """None = no capability metadata for this deployment (e.g. a model absent from the model map, - whose stub info carries no supports_reasoning key at all); () = reasoning unsupported.""" + """None = the caller passed no supports_reasoning key at all; () = this deployment adds no + effort levels to its group. The router always supplies the key, so a deployment absent from the + model map arrives with supports_reasoning None and lands on (), the same answer a mapped + non-reasoning model gets: the group cannot promise a level on behalf of a model nothing is known + about. () therefore reads as "no usable answer" downstream, which is why the dashboard falls + back to the capability-blind level list on an empty group rather than hiding the control.""" if "supports_reasoning" not in model_info: return None if model_info.get("supports_reasoning") is not True: diff --git a/litellm/types/router.py b/litellm/types/router.py index d4c735387a5..fd32b70405b 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -641,6 +641,19 @@ class ModelGroupInfo(BaseModel): supported_openai_params: list[str] | None = Field(default=[]) configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None + @field_validator("supported_reasoning_efforts", mode="before") + @classmethod + def _accept_only_well_formed_reasoning_efforts(cls, value: object) -> tuple[str, ...] | None: + """Deployment model_info reaches this model through a **kwargs splat, so an operator can put + any shape under this key. Anything that is not a level or a list of levels resolves to None + and lets the computed intersection stand, because raising here fails the whole + /model_group/info response rather than the one group that carries the bad value.""" + if isinstance(value, str): + return (value,) + if isinstance(value, (list, tuple)) and all(isinstance(level, str) for level in value): + return tuple(value) + return None + def __init__(self, **data) -> None: for field_name, field_type in get_type_hints(self.__class__).items(): if field_type is bool and data.get(field_name) is None: diff --git a/tests/test_litellm/llms/openai/test_gpt5_transformation.py b/tests/test_litellm/llms/openai/test_gpt5_transformation.py index f2fc7934164..eaae3d8c2c5 100644 --- a/tests/test_litellm/llms/openai/test_gpt5_transformation.py +++ b/tests/test_litellm/llms/openai/test_gpt5_transformation.py @@ -1312,18 +1312,32 @@ def test_responses_gpt54_allow_temperature_effort_none( @pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"]) -def test_gpt5_6_rejects_reasoning_effort_max_on_chat_completions(config: OpenAIConfig, model: str): +def test_gpt5_6_forwards_reasoning_effort_max_for_the_responses_bridge(config: OpenAIConfig, model: str): + """A chat request carrying tools or a reasoning summary is converted to /v1/responses further + down main.py, and that surface accepts max. This runs before litellm has decided to bridge, so + refusing max here would break the cursor thinking-max shape that works today. Plain chat + completions still answer max with a provider 400, and the capability list below is what keeps + the level out of the picker.""" + params = config.map_openai_params( + non_default_params={"reasoning_effort": "max"}, + optional_params={}, + model=model, + drop_params=False, + ) + assert params["reasoning_effort"] == "max" + + +@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"]) +def test_gpt5_6_never_advertises_reasoning_effort_max(model: str): """/v1/chat/completions answers max with "Unsupported value: 'reasoning_effort' does not support - 'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'" on - every gpt-5.6 snapshot, so no gpt-5.6 map entry asserts supports_max_reasoning_effort and the - level is refused here instead of being forwarded into a provider 400.""" - with pytest.raises(litellm.utils.UnsupportedParamsError): - config.map_openai_params( - non_default_params={"reasoning_effort": "max"}, - optional_params={}, - model=model, - drop_params=False, - ) + 'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", so no + gpt-5.6 entry asserts supports_max_reasoning_effort and the advertised set stops at xhigh.""" + from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts + + resolved = resolve_supported_reasoning_efforts(litellm.get_model_info(model)) + assert resolved is not None + assert "max" not in resolved + assert "xhigh" in resolved def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api( @@ -1341,34 +1355,43 @@ def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api( assert params["reasoning"] == {"effort": "max"} -def test_gpt5_6_rejects_reasoning_effort_ultra_until_the_map_opts_in(config: OpenAIConfig): - # ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance - # documents effort values only up to max, and the builder guide frames ultra as multi-agent - # orchestration. A verified supports_ultra_reasoning_effort flag lights this up with no code. - with pytest.raises(litellm.utils.UnsupportedParamsError): - config.map_openai_params( - non_default_params={"reasoning_effort": "ultra"}, - optional_params={}, - model="gpt-5.6", - drop_params=False, - ) +def test_gpt5_6_never_advertises_reasoning_effort_ultra(): + """ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance + documents effort values only up to max, and /v1/responses answers ultra with a 400. A verified + supports_ultra_reasoning_effort flag lights it up with no code change.""" + from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts + + resolved = resolve_supported_reasoning_efforts(litellm.get_model_info("gpt-5.6")) + assert resolved is not None + assert "ultra" not in resolved @pytest.mark.parametrize("effort", ["max", "ultra"]) -def test_gpt5_rejects_opt_in_reasoning_efforts_for_other_models(config: OpenAIConfig, effort: str): +def test_gpt5_forwards_levels_the_chat_gate_does_not_own(config: OpenAIConfig, effort: str): + """Only xhigh is gated on this surface. max and ultra reach the provider (or the responses + bridge) and are answered there, which is what happened before per-group capabilities existed.""" + params = config.map_openai_params( + non_default_params={"reasoning_effort": effort}, + optional_params={}, + model="gpt-5.1", + drop_params=False, + ) + assert params["reasoning_effort"] == effort + + +def test_gpt5_rejects_xhigh_for_models_without_the_flag(config: OpenAIConfig): with pytest.raises(litellm.utils.UnsupportedParamsError): config.map_openai_params( - non_default_params={"reasoning_effort": effort}, + non_default_params={"reasoning_effort": "xhigh"}, optional_params={}, model="gpt-5.1", drop_params=False, ) -@pytest.mark.parametrize("effort", ["max", "ultra"]) -def test_gpt5_drops_opt_in_reasoning_efforts_when_requested(config: OpenAIConfig, effort: str): +def test_gpt5_drops_xhigh_when_requested(config: OpenAIConfig): params = config.map_openai_params( - non_default_params={"reasoning_effort": effort}, + non_default_params={"reasoning_effort": "xhigh"}, optional_params={}, model="gpt-5.1", drop_params=True, diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index 095c962328a..62c6411c3cc 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -8927,7 +8927,11 @@ def test_model_group_info_intersects_supported_reasoning_efforts(): assert result.supported_reasoning_efforts == ("minimal", "low", "medium", "high") -def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata(): +def test_model_group_info_reasoning_efforts_empty_when_a_deployment_is_off_the_map(): + """The router fills every ModelInfo key, so a deployment absent from the model map arrives with + supports_reasoning None rather than with the key missing, and the group can no longer promise a + level on its behalf. The dashboard reads the empty result as "nothing known" and falls back to + the capability-blind picker, which is what it showed before this field existed.""" router = litellm.Router( model_list=[ { @@ -8952,7 +8956,7 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata( "supports_reasoning": True, "supports_max_reasoning_effort": True, } - return {"key": model_name, "litellm_provider": "openai", "mode": "chat"} + return {"key": model_name, "litellm_provider": "openai", "mode": None, "supports_reasoning": None} with patch.object(router, "get_deployment_model_info", side_effect=_model_info): result = router._set_model_group_info( @@ -8961,4 +8965,29 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata( ) assert result is not None - assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high", "max") + assert result.supported_reasoning_efforts == () + + +@pytest.mark.parametrize( + "configured, expected", + [ + ("high", ("high",)), + (["low", "high"], ("low", "high")), + (17, None), + ({"effort": "high"}, None), + ([1, 2], None), + ], +) +def test_model_group_info_tolerates_any_configured_reasoning_efforts_shape(configured, expected): + """Deployment model_info is splatted into ModelGroupInfo, so this key arrives with whatever an + operator wrote in the config. A shape pydantic cannot validate used to fail the whole + /model_group/info response, not just the group carrying it.""" + from litellm.types.router import ModelGroupInfo + + info = ModelGroupInfo( + model_group="g", + providers=["openai"], + supported_reasoning_efforts=configured, + ) + + assert info.supported_reasoning_efforts == expected diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx index a26197aa243..09a710c8729 100644 --- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx @@ -968,6 +968,24 @@ describe("ComplexityRouterConfig per-model effort filtering", () => { expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]); }); + it("falls back to every effort when the group intersects to nothing", async () => { + renderWithProviders( + model.model_group !== "claude-3-opus"), + { model_group: "claude-3-opus", mode: "chat", supports_reasoning: true, supported_reasoning_efforts: [] }, + ]} + />, + ); + const user = userEvent.setup(); + await user.click( + screen.getByRole("combobox", { name: "Reasoning effort for claude-3-opus in the Reasoning tier" }), + ); + const options = (await screen.findAllByRole("option")).map((option) => option.textContent); + expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]); + }); + // Hand-authored configs can carry a level outside the supported set (e.g. max); it must render // and stay clearable rather than being masked as Default. it("keeps showing a stored effort outside the supported set", () => { diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx index e72905c2f73..77e093b63d9 100644 --- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx @@ -226,6 +226,20 @@ export const effectiveTierLabel = (tier: keyof ComplexityTiers, tierLabels: Comp export const planModeEligibleTiers = (tiers: ComplexityTiers): Array => TIER_KEYS.filter((tier) => (tiers[tier] ?? []).length > 0); +/** + * The backend list is the per-group intersection of accepted effort levels. An empty intersection + * carries no more information than a proxy that does not send the field yet (one deployment absent + * from the model map empties it), so both fall back to the coarse supports_reasoning gate with every + * level offered, which is what this dropdown showed before the field existed. + */ +const effortOptionsForModel = (model: ModelGroup): string[] => { + const advertised = model.supported_reasoning_efforts ?? []; + if (advertised.length > 0) { + return [...advertised]; + } + return model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : []; +}; + const ComplexityRouterConfig: React.FC = ({ modelInfo, value, @@ -251,16 +265,11 @@ const ComplexityRouterConfig: React.FC = ({ const derivedDefaultModel = resolveComplexityDefaultModel(value.tiers); const defaultModel = resolveComplexityDefaultModel(value.tiers, value.default_model); - // Embedding models can't serve a chat-completion role, so they're excluded here. - // The backend list is the per-group intersection of accepted effort levels; when a proxy does not - // send it yet, fall back to the coarse supports_reasoning gate with every level offered. const effortOptionsByModel: Record = Object.fromEntries( - modelInfo.map((model) => [ - model.model_group, - model.supported_reasoning_efforts ?? (model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : []), - ]), + modelInfo.map((model) => [model.model_group, effortOptionsForModel(model)]), ); + // Embedding models can't serve a chat-completion role, so they're excluded here. const modelOptions = modelInfo .filter((model) => model.mode !== "embedding") .map((model) => ({