diff --git a/litellm/router.py b/litellm/router.py index 8cd1b804928..bd5f97b27c4 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -9485,6 +9485,7 @@ class Router: "model_group": user_facing_model_group_name, "providers": [llm_provider], **model_info, + "supported_reasoning_efforts": None, } ) else: diff --git a/litellm/router_utils/reasoning_effort_capability.py b/litellm/router_utils/reasoning_effort_capability.py index 53a159af7d1..d445d611cd6 100644 --- a/litellm/router_utils/reasoning_effort_capability.py +++ b/litellm/router_utils/reasoning_effort_capability.py @@ -1,11 +1,18 @@ """Resolve which reasoning_effort values a deployment, and by intersection a model group, accepts. The model-map flags carry different polarity per level, mirroring the provider gates -(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out; -anthropic/chat/transformation.py rejects only xhigh/max without an explicit flag): medium and high -are unconditional for any reasoning model, minimal/low are supported unless the map explicitly -says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that polarity -in one place instead of re-encoding it in every consumer. +(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out): +medium and high are unconditional for any reasoning model, minimal/low are supported unless the map +explicitly says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that +polarity in one place instead of re-encoding it in every consumer. + +Only openai and azure gate xhigh and max on the request path. anthropic/chat/transformation.py +gates them on the output_config path alone, so its reasoning_effort path maps every level to a +thinking budget whatever the map says, and a claude group whose entry omits +supports_xhigh_reasoning_effort stops offering a level litellm would have forwarded. That is the +deliberate trade: an explicit flag is the only signal that the tier is a real one rather than +litellm quietly rounding the level to a budget, and what gets dropped is advisory metadata rather +than a restriction on the request path. The none level is the one flag whose polarity is provider-dependent. OpenAI never refuses it on the request path (azure/chat/gpt_5_transformation.py is the only caller that does, and it raises @@ -58,9 +65,15 @@ def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tup known model that accepts no effort level, which correctly empties the group. Keeping the two apart matters because the router registers a deployment absent from the model map under a synthesized entry, and get_model_info then answers with supports_reasoning None exactly as it - does for a mapped non-reasoning model. That entry carries no mode, which every real map entry for - a routable model does, so an unset flag with no mode is the unknown case and one custom model in - a group no longer wipes the levels its mapped siblings agree on.""" + does for a mapped non-reasoning model. Mode is what separates them: every map entry for a + routable model declares one, so an unset flag with no mode is read as unknown and one custom + model in a group no longer wipes the levels its mapped siblings agree on. + + The separation is only as good as that signal. An off-map deployment carrying any model_info of + its own is registered under its deployment id with mode defaulting to chat, so it reads as a + known non-reasoning model and does still empty its group, with supports_reasoning on that + deployment as the way out. Telling the two apart for real needs provenance that the flattened + ModelInfo does not carry.""" if "supports_reasoning" not in model_info: return None supports_reasoning: Final = model_info.get("supports_reasoning") diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index ba23075e7cc..b8a5a454613 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -9007,6 +9007,46 @@ def test_model_group_info_reasoning_efforts_empty_on_a_mapped_non_reasoning_depl assert result.supported_reasoning_efforts == () +def test_model_group_info_reasoning_efforts_ignore_a_value_declared_in_model_info(): + """The group's levels are computed from its deployments, so a value an operator left in one + deployment's model_info must not seed them. Seeding let the first deployment read narrow the + whole group while the same value on any other deployment was silently ignored.""" + router = litellm.Router( + model_list=[ + { + "model_name": "declared-group", + "litellm_params": {"model": "openai/first-reasoner"}, + "model_info": {"id": "first-deployment"}, + }, + { + "model_name": "declared-group", + "litellm_params": {"model": "openai/second-reasoner"}, + "model_info": {"id": "second-deployment"}, + }, + ] + ) + + def _model_info(model_id: str, model_name: str): + info = { + "key": model_name, + "litellm_provider": "openai", + "mode": "chat", + "supports_reasoning": True, + } + if model_id == "first-deployment": + info["supported_reasoning_efforts"] = ("high",) + return info + + with patch.object(router, "get_deployment_model_info", side_effect=_model_info): + result = router._set_model_group_info( + model_group="declared-group", + user_facing_model_group_name="declared-group", + ) + + assert result is not None + assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high") + + @pytest.mark.parametrize( "configured, expected", [