mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(router): fix least-busy routing always picking the same deployment when counts are tied
When multiple deployments had equal in-flight request counts (e.g. all zero on startup, or naturally converging), the selection loop used a strict less-than comparison and always returned the first deployment encountered in dict iteration order. This caused all traffic to pile on one node while others idled. Changes: - Collect all deployments tied at the minimum count into a list and randomly pick from them, so load is spread evenly when counts are equal - Skip stale cache entries (deployments no longer in the healthy list) when scanning the count dict, preventing phantom entries from influencing routing decisions - Add unit tests covering equal-count distribution, unique-minimum selection, stale-entry filtering, and sync/async cache paths Made-with: Cursor
This commit is contained in:
parent
ef9cc33ee3
commit
2f2e0cf0a8
2 changed files with 142 additions and 14 deletions
|
|
@ -197,27 +197,31 @@ class LeastBusyLoggingHandler(CustomLogger):
|
|||
"""
|
||||
Helper to get deployments using least busy strategy
|
||||
"""
|
||||
healthy_ids = set()
|
||||
for d in healthy_deployments:
|
||||
## if healthy deployment not yet used
|
||||
if d["model_info"]["id"] not in all_deployments:
|
||||
all_deployments[d["model_info"]["id"]] = 0
|
||||
# map deployment to id
|
||||
# pick least busy deployment
|
||||
_id = d["model_info"]["id"]
|
||||
healthy_ids.add(_id)
|
||||
if _id not in all_deployments:
|
||||
all_deployments[_id] = 0
|
||||
|
||||
min_traffic = float("inf")
|
||||
min_deployment = None
|
||||
min_deployment_ids: list = []
|
||||
for k, v in all_deployments.items():
|
||||
if k not in healthy_ids:
|
||||
continue
|
||||
if v < min_traffic:
|
||||
min_traffic = v
|
||||
min_deployment = k
|
||||
if min_deployment is not None:
|
||||
## check if min deployment is a string, if so, cast it to int
|
||||
min_deployment_ids = [k]
|
||||
elif v == min_traffic:
|
||||
min_deployment_ids.append(k)
|
||||
|
||||
if min_deployment_ids:
|
||||
chosen_id = random.choice(min_deployment_ids)
|
||||
for m in healthy_deployments:
|
||||
if m["model_info"]["id"] == min_deployment:
|
||||
if m["model_info"]["id"] == chosen_id:
|
||||
return m
|
||||
min_deployment = random.choice(healthy_deployments)
|
||||
else:
|
||||
min_deployment = random.choice(healthy_deployments)
|
||||
return min_deployment
|
||||
|
||||
return random.choice(healthy_deployments)
|
||||
|
||||
def get_available_deployments(
|
||||
self,
|
||||
|
|
|
|||
124
tests/test_litellm/router_strategy/test_least_busy.py
Normal file
124
tests/test_litellm/router_strategy/test_least_busy.py
Normal file
|
|
@ -0,0 +1,124 @@
|
|||
import os
|
||||
import sys
|
||||
from collections import Counter
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.router_strategy.least_busy import LeastBusyLoggingHandler
|
||||
|
||||
|
||||
def _make_deployment(id_val):
|
||||
return {
|
||||
"model_name": "test-model",
|
||||
"litellm_params": {"model": "openai/gpt-4.1-mini"},
|
||||
"model_info": {"id": str(id_val)},
|
||||
}
|
||||
|
||||
|
||||
class TestLeastBusyTieBreaking:
|
||||
"""Tests that least-busy strategy distributes requests across tied deployments."""
|
||||
|
||||
def test_should_randomly_distribute_when_all_counts_are_zero(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
deployments = [_make_deployment(i) for i in range(3)]
|
||||
|
||||
selections = Counter()
|
||||
for _ in range(300):
|
||||
chosen = handler._get_available_deployments(
|
||||
healthy_deployments=deployments, all_deployments={}
|
||||
)
|
||||
selections[chosen["model_info"]["id"]] += 1
|
||||
|
||||
assert len(selections) == 3, (
|
||||
f"Expected all 3 deployments to be selected at least once, got: {selections}"
|
||||
)
|
||||
for dep_id, count in selections.items():
|
||||
assert count > 30, (
|
||||
f"Deployment {dep_id} selected only {count}/300 times — distribution is too skewed"
|
||||
)
|
||||
|
||||
def test_should_randomly_distribute_when_counts_are_tied(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
deployments = [_make_deployment(i) for i in range(2)]
|
||||
tied_counts = {"0": 5, "1": 5}
|
||||
|
||||
selections = Counter()
|
||||
for _ in range(200):
|
||||
chosen = handler._get_available_deployments(
|
||||
healthy_deployments=deployments,
|
||||
all_deployments=dict(tied_counts),
|
||||
)
|
||||
selections[chosen["model_info"]["id"]] += 1
|
||||
|
||||
assert len(selections) == 2, (
|
||||
f"Expected both deployments to be selected, got: {selections}"
|
||||
)
|
||||
for dep_id, count in selections.items():
|
||||
assert count > 40, (
|
||||
f"Deployment {dep_id} selected only {count}/200 times — distribution is too skewed"
|
||||
)
|
||||
|
||||
def test_should_pick_unique_minimum(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
deployments = [_make_deployment(i) for i in range(3)]
|
||||
counts = {"0": 10, "1": 1, "2": 50}
|
||||
|
||||
for _ in range(50):
|
||||
chosen = handler._get_available_deployments(
|
||||
healthy_deployments=deployments,
|
||||
all_deployments=dict(counts),
|
||||
)
|
||||
assert chosen["model_info"]["id"] == "1"
|
||||
|
||||
def test_should_skip_unhealthy_deployments_in_cache(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
healthy = [_make_deployment(1), _make_deployment(2)]
|
||||
counts_with_stale = {"0": 0, "1": 5, "2": 10}
|
||||
|
||||
for _ in range(50):
|
||||
chosen = handler._get_available_deployments(
|
||||
healthy_deployments=healthy,
|
||||
all_deployments=dict(counts_with_stale),
|
||||
)
|
||||
assert chosen["model_info"]["id"] in ("1", "2")
|
||||
assert chosen["model_info"]["id"] == "1"
|
||||
|
||||
|
||||
class TestLeastBusyGetAvailableDeployments:
|
||||
"""Tests the sync/async wrappers for cache interaction."""
|
||||
|
||||
def test_should_use_cache_for_selection(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
deployments = [_make_deployment(0), _make_deployment(1)]
|
||||
|
||||
cache.set_cache(
|
||||
key="test-model_request_count", value={"0": 10, "1": 2}
|
||||
)
|
||||
chosen = handler.get_available_deployments(
|
||||
model_group="test-model", healthy_deployments=deployments
|
||||
)
|
||||
assert chosen["model_info"]["id"] == "1"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_should_use_cache_for_async_selection(self):
|
||||
cache = DualCache()
|
||||
handler = LeastBusyLoggingHandler(router_cache=cache)
|
||||
deployments = [_make_deployment(0), _make_deployment(1)]
|
||||
|
||||
await cache.async_set_cache(
|
||||
key="test-model_request_count", value={"0": 10, "1": 2}
|
||||
)
|
||||
chosen = await handler.async_get_available_deployments(
|
||||
model_group="test-model", healthy_deployments=deployments
|
||||
)
|
||||
assert chosen["model_info"]["id"] == "1"
|
||||
Loading…
Add table
Reference in a new issue