fix(router): fix least-busy routing always picking the same deployment when counts are tied

When multiple deployments had equal in-flight request counts (e.g. all
zero on startup, or naturally converging), the selection loop used a
strict less-than comparison and always returned the first deployment
encountered in dict iteration order. This caused all traffic to pile on
one node while others idled.

Changes:
- Collect all deployments tied at the minimum count into a list and
  randomly pick from them, so load is spread evenly when counts are equal
- Skip stale cache entries (deployments no longer in the healthy list)
  when scanning the count dict, preventing phantom entries from
  influencing routing decisions
- Add unit tests covering equal-count distribution, unique-minimum
  selection, stale-entry filtering, and sync/async cache paths

Made-with: Cursor
This commit is contained in:
CJYLZS 2026-03-18 08:35:17 +08:00
parent ef9cc33ee3
commit 2f2e0cf0a8
2 changed files with 142 additions and 14 deletions

View file

@ -197,27 +197,31 @@ class LeastBusyLoggingHandler(CustomLogger):
"""
Helper to get deployments using least busy strategy
"""
healthy_ids = set()
for d in healthy_deployments:
## if healthy deployment not yet used
if d["model_info"]["id"] not in all_deployments:
all_deployments[d["model_info"]["id"]] = 0
# map deployment to id
# pick least busy deployment
_id = d["model_info"]["id"]
healthy_ids.add(_id)
if _id not in all_deployments:
all_deployments[_id] = 0
min_traffic = float("inf")
min_deployment = None
min_deployment_ids: list = []
for k, v in all_deployments.items():
if k not in healthy_ids:
continue
if v < min_traffic:
min_traffic = v
min_deployment = k
if min_deployment is not None:
## check if min deployment is a string, if so, cast it to int
min_deployment_ids = [k]
elif v == min_traffic:
min_deployment_ids.append(k)
if min_deployment_ids:
chosen_id = random.choice(min_deployment_ids)
for m in healthy_deployments:
if m["model_info"]["id"] == min_deployment:
if m["model_info"]["id"] == chosen_id:
return m
min_deployment = random.choice(healthy_deployments)
else:
min_deployment = random.choice(healthy_deployments)
return min_deployment
return random.choice(healthy_deployments)
def get_available_deployments(
self,

View file

@ -0,0 +1,124 @@
import os
import sys
from collections import Counter
import pytest
sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system path
from litellm.caching.caching import DualCache
from litellm.router_strategy.least_busy import LeastBusyLoggingHandler
def _make_deployment(id_val):
return {
"model_name": "test-model",
"litellm_params": {"model": "openai/gpt-4.1-mini"},
"model_info": {"id": str(id_val)},
}
class TestLeastBusyTieBreaking:
"""Tests that least-busy strategy distributes requests across tied deployments."""
def test_should_randomly_distribute_when_all_counts_are_zero(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
deployments = [_make_deployment(i) for i in range(3)]
selections = Counter()
for _ in range(300):
chosen = handler._get_available_deployments(
healthy_deployments=deployments, all_deployments={}
)
selections[chosen["model_info"]["id"]] += 1
assert len(selections) == 3, (
f"Expected all 3 deployments to be selected at least once, got: {selections}"
)
for dep_id, count in selections.items():
assert count > 30, (
f"Deployment {dep_id} selected only {count}/300 times — distribution is too skewed"
)
def test_should_randomly_distribute_when_counts_are_tied(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
deployments = [_make_deployment(i) for i in range(2)]
tied_counts = {"0": 5, "1": 5}
selections = Counter()
for _ in range(200):
chosen = handler._get_available_deployments(
healthy_deployments=deployments,
all_deployments=dict(tied_counts),
)
selections[chosen["model_info"]["id"]] += 1
assert len(selections) == 2, (
f"Expected both deployments to be selected, got: {selections}"
)
for dep_id, count in selections.items():
assert count > 40, (
f"Deployment {dep_id} selected only {count}/200 times — distribution is too skewed"
)
def test_should_pick_unique_minimum(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
deployments = [_make_deployment(i) for i in range(3)]
counts = {"0": 10, "1": 1, "2": 50}
for _ in range(50):
chosen = handler._get_available_deployments(
healthy_deployments=deployments,
all_deployments=dict(counts),
)
assert chosen["model_info"]["id"] == "1"
def test_should_skip_unhealthy_deployments_in_cache(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
healthy = [_make_deployment(1), _make_deployment(2)]
counts_with_stale = {"0": 0, "1": 5, "2": 10}
for _ in range(50):
chosen = handler._get_available_deployments(
healthy_deployments=healthy,
all_deployments=dict(counts_with_stale),
)
assert chosen["model_info"]["id"] in ("1", "2")
assert chosen["model_info"]["id"] == "1"
class TestLeastBusyGetAvailableDeployments:
"""Tests the sync/async wrappers for cache interaction."""
def test_should_use_cache_for_selection(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
deployments = [_make_deployment(0), _make_deployment(1)]
cache.set_cache(
key="test-model_request_count", value={"0": 10, "1": 2}
)
chosen = handler.get_available_deployments(
model_group="test-model", healthy_deployments=deployments
)
assert chosen["model_info"]["id"] == "1"
@pytest.mark.asyncio
async def test_should_use_cache_for_async_selection(self):
cache = DualCache()
handler = LeastBusyLoggingHandler(router_cache=cache)
deployments = [_make_deployment(0), _make_deployment(1)]
await cache.async_set_cache(
key="test-model_request_count", value={"0": 10, "1": 2}
)
chosen = await handler.async_get_available_deployments(
model_group="test-model", healthy_deployments=deployments
)
assert chosen["model_info"]["id"] == "1"