mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
Merge pull request #40991 from BerriAI/litellm_team_model_cooldown_siblings
fix(router): cool down team deployments on 429 when a sibling serves the same public model
This commit is contained in:
commit
e766277846
4 changed files with 119 additions and 2 deletions
|
|
@ -1538,6 +1538,18 @@ class Router:
|
|||
return False
|
||||
return sum(len(self.model_name_to_deployment_indices.get(member) or ()) for member in group.models) > 1
|
||||
|
||||
def team_model_has_alternatives(self, deployment_id: str) -> bool:
|
||||
deployment: Final = self.get_deployment(model_id=deployment_id)
|
||||
if deployment is None:
|
||||
return False
|
||||
team_id: Final = deployment.model_info.team_id
|
||||
public_model_name: Final = deployment.model_info.team_public_model_name
|
||||
if team_id is None or public_model_name is None:
|
||||
return False
|
||||
sibling_indices: Final = self.team_model_to_deployment_indices.get((team_id, public_model_name)) or ()
|
||||
routable_siblings: Final = self._filter_blocked_deployments([self.model_list[idx] for idx in sibling_indices])
|
||||
return len(routable_siblings) > 1
|
||||
|
||||
_OVERRIDABLE_ROUTING_STRATEGIES: frozenset[str] = frozenset({"simple-shuffle", *_DEFAULT_SELECTOR_ATTR_BY_STRATEGY})
|
||||
|
||||
def _get_request_routing_strategy_override(self, request_kwargs: dict | None) -> str | None:
|
||||
|
|
|
|||
|
|
@ -343,8 +343,9 @@ def _should_cooldown_deployment(
|
|||
model_group: Final = litellm_router_instance.get_model_group(id=deployment)
|
||||
is_single_deployment_model_group = False
|
||||
if model_group is not None and len(model_group) == 1:
|
||||
is_single_deployment_model_group = not litellm_router_instance.routing_group_has_alternatives(
|
||||
requested_model_group
|
||||
is_single_deployment_model_group = not (
|
||||
litellm_router_instance.routing_group_has_alternatives(requested_model_group)
|
||||
or litellm_router_instance.team_model_has_alternatives(deployment)
|
||||
)
|
||||
|
||||
## CHECK DEPLOYMENT-LEVEL POLICY FIRST (overrides router-level)
|
||||
|
|
|
|||
|
|
@ -437,3 +437,67 @@ class TestRoutingGroupCooldownAlternatives:
|
|||
)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
class TestTeamModelCooldownAlternatives:
|
||||
def _router(self, team_deployments: int, blocked_ids: frozenset[str] = frozenset()) -> litellm.Router:
|
||||
return litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": f"model_name_team-1_{i}",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
|
||||
"model_info": {
|
||||
"id": f"team-deploy-{i}",
|
||||
"team_id": "team-1",
|
||||
"team_public_model_name": "team-gpt-4o-mini",
|
||||
"blocked": f"team-deploy-{i}" in blocked_ids,
|
||||
},
|
||||
}
|
||||
for i in range(team_deployments)
|
||||
]
|
||||
)
|
||||
|
||||
def test_429_on_team_deployment_with_sibling_cools_down(self):
|
||||
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
|
||||
|
||||
router = self._router(team_deployments=2)
|
||||
assert (
|
||||
_should_cooldown_deployment(
|
||||
litellm_router_instance=router,
|
||||
deployment="team-deploy-0",
|
||||
exception_status=429,
|
||||
original_exception=Exception("rate limited"),
|
||||
requested_model_group="team-gpt-4o-mini",
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
def test_429_on_only_team_deployment_keeps_single_deployment_exemption(self):
|
||||
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
|
||||
|
||||
router = self._router(team_deployments=1)
|
||||
assert (
|
||||
_should_cooldown_deployment(
|
||||
litellm_router_instance=router,
|
||||
deployment="team-deploy-0",
|
||||
exception_status=429,
|
||||
original_exception=Exception("rate limited"),
|
||||
requested_model_group="team-gpt-4o-mini",
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
||||
def test_429_with_only_a_blocked_sibling_keeps_single_deployment_exemption(self):
|
||||
from litellm.router_utils.cooldown_handlers import _should_cooldown_deployment
|
||||
|
||||
router = self._router(team_deployments=2, blocked_ids=frozenset({"team-deploy-1"}))
|
||||
assert (
|
||||
_should_cooldown_deployment(
|
||||
litellm_router_instance=router,
|
||||
deployment="team-deploy-0",
|
||||
exception_status=429,
|
||||
original_exception=Exception("rate limited"),
|
||||
requested_model_group="team-gpt-4o-mini",
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
|
|
|||
|
|
@ -896,6 +896,46 @@ def test_arouter_test_team_model():
|
|||
assert result is not None
|
||||
|
||||
|
||||
def test_team_model_has_alternatives():
|
||||
def team_deployment(
|
||||
deployment_id: str, team_id: str, public_model_name: str, blocked: bool = False
|
||||
) -> DeploymentTypedDict:
|
||||
return {
|
||||
"model_name": f"model_name_{team_id}_{deployment_id}",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
|
||||
"model_info": {
|
||||
"id": deployment_id,
|
||||
"team_id": team_id,
|
||||
"team_public_model_name": public_model_name,
|
||||
"blocked": blocked,
|
||||
},
|
||||
}
|
||||
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
team_deployment("team-a-1", "team-a", "shared-model"),
|
||||
team_deployment("team-a-2", "team-a", "shared-model"),
|
||||
team_deployment("team-a-solo", "team-a", "solo-model"),
|
||||
team_deployment("team-b-1", "team-b", "shared-model"),
|
||||
team_deployment("team-c-1", "team-c", "paused-sibling-model"),
|
||||
team_deployment("team-c-paused", "team-c", "paused-sibling-model", blocked=True),
|
||||
{
|
||||
"model_name": "plain-model",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"},
|
||||
"model_info": {"id": "plain-1"},
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
assert router.team_model_has_alternatives("team-a-1") is True
|
||||
assert router.team_model_has_alternatives("team-a-2") is True
|
||||
assert router.team_model_has_alternatives("team-a-solo") is False
|
||||
assert router.team_model_has_alternatives("team-b-1") is False
|
||||
assert router.team_model_has_alternatives("team-c-1") is False
|
||||
assert router.team_model_has_alternatives("plain-1") is False
|
||||
assert router.team_model_has_alternatives("missing-deployment") is False
|
||||
|
||||
|
||||
def test_arouter_ignore_invalid_deployments():
|
||||
"""
|
||||
Test that router.ignore_invalid_deployments is set to True
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue