Merge pull request #40991 from BerriAI/litellm_team_model_cooldown_siblings

fix(router): cool down team deployments on 429 when a sibling serves the same public model
This commit is contained in:
Yassin Kortam 2026-09-14 14:57:07 -07:00 • committed by GitHub
commit e766277846
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 119 additions and 2 deletions

View file

@ -1538,6 +1538,18 @@ class Router:
return False
return sum(len(self.model_name_to_deployment_indices.get(member) or ()) for member in group.models) > 1
def team_model_has_alternatives(self, deployment_id: str) -> bool:
deployment: Final = self.get_deployment(model_id=deployment_id)
if deployment is None:
return False
team_id: Final = deployment.model_info.team_id
public_model_name: Final = deployment.model_info.team_public_model_name
if team_id is None or public_model_name is None:
return False
sibling_indices: Final = self.team_model_to_deployment_indices.get((team_id, public_model_name)) or ()
routable_siblings: Final = self._filter_blocked_deployments([self.model_list[idx] for idx in sibling_indices])
return len(routable_siblings) > 1
_OVERRIDABLE_ROUTING_STRATEGIES: frozenset[str] = frozenset({"simple-shuffle", *_DEFAULT_SELECTOR_ATTR_BY_STRATEGY})
def _get_request_routing_strategy_override(self, request_kwargs: dict | None) -> str | None:

View file

@ -343,8 +343,9 @@ def _should_cooldown_deployment(
model_group: Final = litellm_router_instance.get_model_group(id=deployment)
is_single_deployment_model_group = False
if model_group is not None and len(model_group) == 1:
is_single_deployment_model_group = not litellm_router_instance.routing_group_has_alternatives(
requested_model_group
is_single_deployment_model_group = not (
litellm_router_instance.routing_group_has_alternatives(requested_model_group)
or litellm_router_instance.team_model_has_alternatives(deployment)
)
## CHECK DEPLOYMENT-LEVEL POLICY FIRST (overrides router-level)

View file

@ -437,3 +437,67 @@ class TestRoutingGroupCooldownAlternatives:
)
is False
)
class TestTeamModelCooldownAlternatives:
def _router(self, team_deployments: int, blocked_ids: frozenset[str] = frozenset()) -> litellm.Router:
return litellm.Router(
model_list=[
{
"model_name": f"model_name_team-1_{i}",
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
"model_info": {
"id": f"team-deploy-{i}",
"team_id": "team-1",
"team_public_model_name": "team-gpt-4o-mini",
"blocked": f"team-deploy-{i}" in blocked_ids,
},
}
for i in range(team_deployments)
]
)
def test_429_on_team_deployment_with_sibling_cools_down(self):
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
router = self._router(team_deployments=2)
assert (
_should_cooldown_deployment(
litellm_router_instance=router,
deployment="team-deploy-0",
exception_status=429,
original_exception=Exception("rate limited"),
requested_model_group="team-gpt-4o-mini",
)
is True
)
def test_429_on_only_team_deployment_keeps_single_deployment_exemption(self):
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
router = self._router(team_deployments=1)
assert (
_should_cooldown_deployment(
litellm_router_instance=router,
deployment="team-deploy-0",
exception_status=429,
original_exception=Exception("rate limited"),
requested_model_group="team-gpt-4o-mini",
)
is False
)
def test_429_with_only_a_blocked_sibling_keeps_single_deployment_exemption(self):
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
router = self._router(team_deployments=2, blocked_ids=frozenset({"team-deploy-1"}))
assert (
_should_cooldown_deployment(
litellm_router_instance=router,
deployment="team-deploy-0",
exception_status=429,
original_exception=Exception("rate limited"),
requested_model_group="team-gpt-4o-mini",
)
is False
)

View file

@ -896,6 +896,46 @@ def test_arouter_test_team_model():
assert result is not None
def test_team_model_has_alternatives():
def team_deployment(
deployment_id: str, team_id: str, public_model_name: str, blocked: bool = False
) -> DeploymentTypedDict:
return {
"model_name": f"model_name_{team_id}_{deployment_id}",
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
"model_info": {
"id": deployment_id,
"team_id": team_id,
"team_public_model_name": public_model_name,
"blocked": blocked,
},
}
router = litellm.Router(
model_list=[
team_deployment("team-a-1", "team-a", "shared-model"),
team_deployment("team-a-2", "team-a", "shared-model"),
team_deployment("team-a-solo", "team-a", "solo-model"),
team_deployment("team-b-1", "team-b", "shared-model"),
team_deployment("team-c-1", "team-c", "paused-sibling-model"),
team_deployment("team-c-paused", "team-c", "paused-sibling-model", blocked=True),
{
"model_name": "plain-model",
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
"model_info": {"id": "plain-1"},
},
],
)
assert router.team_model_has_alternatives("team-a-1") is True
assert router.team_model_has_alternatives("team-a-2") is True
assert router.team_model_has_alternatives("team-a-solo") is False
assert router.team_model_has_alternatives("team-b-1") is False
assert router.team_model_has_alternatives("team-c-1") is False
assert router.team_model_has_alternatives("plain-1") is False
assert router.team_model_has_alternatives("missing-deployment") is False
def test_arouter_ignore_invalid_deployments():
"""
Test that router.ignore_invalid_deployments is set to True