mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(router): always compute a group's reasoning efforts rather than seeding them from model_info
The ModelGroupInfo splat let a supported_reasoning_efforts value left in a deployment's model_info seed the group, so it narrowed the group from whichever deployment was read first and was silently ignored on every other one. The field is derived from the group's deployments, so start it unset and let the intersection fill it in. Also correct two docstring claims that did not match the code: the anthropic chat path gates xhigh and max on the output_config path only, and the mode signal separates an unknown deployment from a known non-reasoning one only while that deployment carries no model_info of its own.
This commit is contained in:
parent
96acac1f91
commit
6648ebe3cc
3 changed files with 62 additions and 8 deletions
|
|
@ -9485,6 +9485,7 @@ class Router:
|
|||
"model_group": user_facing_model_group_name,
|
||||
"providers": [llm_provider],
|
||||
**model_info,
|
||||
"supported_reasoning_efforts": None,
|
||||
}
|
||||
)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -1,11 +1,18 @@
|
|||
"""Resolve which reasoning_effort values a deployment, and by intersection a model group, accepts.
|
||||
|
||||
The model-map flags carry different polarity per level, mirroring the provider gates
|
||||
(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out;
|
||||
anthropic/chat/transformation.py rejects only xhigh/max without an explicit flag): medium and high
|
||||
are unconditional for any reasoning model, minimal/low are supported unless the map explicitly
|
||||
says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that polarity
|
||||
in one place instead of re-encoding it in every consumer.
|
||||
(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out):
|
||||
medium and high are unconditional for any reasoning model, minimal/low are supported unless the map
|
||||
explicitly says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that
|
||||
polarity in one place instead of re-encoding it in every consumer.
|
||||
|
||||
Only openai and azure gate xhigh and max on the request path. anthropic/chat/transformation.py
|
||||
gates them on the output_config path alone, so its reasoning_effort path maps every level to a
|
||||
thinking budget whatever the map says, and a claude group whose entry omits
|
||||
supports_xhigh_reasoning_effort stops offering a level litellm would have forwarded. That is the
|
||||
deliberate trade: an explicit flag is the only signal that the tier is a real one rather than
|
||||
litellm quietly rounding the level to a budget, and what gets dropped is advisory metadata rather
|
||||
than a restriction on the request path.
|
||||
|
||||
The none level is the one flag whose polarity is provider-dependent. OpenAI never refuses it on the
|
||||
request path (azure/chat/gpt_5_transformation.py is the only caller that does, and it raises
|
||||
|
|
@ -58,9 +65,15 @@ def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tup
|
|||
known model that accepts no effort level, which correctly empties the group. Keeping the two
|
||||
apart matters because the router registers a deployment absent from the model map under a
|
||||
synthesized entry, and get_model_info then answers with supports_reasoning None exactly as it
|
||||
does for a mapped non-reasoning model. That entry carries no mode, which every real map entry for
|
||||
a routable model does, so an unset flag with no mode is the unknown case and one custom model in
|
||||
a group no longer wipes the levels its mapped siblings agree on."""
|
||||
does for a mapped non-reasoning model. Mode is what separates them: every map entry for a
|
||||
routable model declares one, so an unset flag with no mode is read as unknown and one custom
|
||||
model in a group no longer wipes the levels its mapped siblings agree on.
|
||||
|
||||
The separation is only as good as that signal. An off-map deployment carrying any model_info of
|
||||
its own is registered under its deployment id with mode defaulting to chat, so it reads as a
|
||||
known non-reasoning model and does still empty its group, with supports_reasoning on that
|
||||
deployment as the way out. Telling the two apart for real needs provenance that the flattened
|
||||
ModelInfo does not carry."""
|
||||
if "supports_reasoning" not in model_info:
|
||||
return None
|
||||
supports_reasoning: Final = model_info.get("supports_reasoning")
|
||||
|
|
|
|||
|
|
@ -9007,6 +9007,46 @@ def test_model_group_info_reasoning_efforts_empty_on_a_mapped_non_reasoning_depl
|
|||
assert result.supported_reasoning_efforts == ()
|
||||
|
||||
|
||||
def test_model_group_info_reasoning_efforts_ignore_a_value_declared_in_model_info():
|
||||
"""The group's levels are computed from its deployments, so a value an operator left in one
|
||||
deployment's model_info must not seed them. Seeding let the first deployment read narrow the
|
||||
whole group while the same value on any other deployment was silently ignored."""
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "declared-group",
|
||||
"litellm_params": {"model": "openai/first-reasoner"},
|
||||
"model_info": {"id": "first-deployment"},
|
||||
},
|
||||
{
|
||||
"model_name": "declared-group",
|
||||
"litellm_params": {"model": "openai/second-reasoner"},
|
||||
"model_info": {"id": "second-deployment"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
def _model_info(model_id: str, model_name: str):
|
||||
info = {
|
||||
"key": model_name,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
"supports_reasoning": True,
|
||||
}
|
||||
if model_id == "first-deployment":
|
||||
info["supported_reasoning_efforts"] = ("high",)
|
||||
return info
|
||||
|
||||
with patch.object(router, "get_deployment_model_info", side_effect=_model_info):
|
||||
result = router._set_model_group_info(
|
||||
model_group="declared-group",
|
||||
user_facing_model_group_name="declared-group",
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"configured, expected",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue