From 2f2e0cf0a8f4c6dd7dc25f8e2586b7959cf61897 Mon Sep 17 00:00:00 2001 From: CJYLZS <691086891@qq.com> Date: Wed, 18 Mar 2026 08:35:17 +0800 Subject: [PATCH] fix(router): fix least-busy routing always picking the same deployment when counts are tied When multiple deployments had equal in-flight request counts (e.g. all zero on startup, or naturally converging), the selection loop used a strict less-than comparison and always returned the first deployment encountered in dict iteration order. This caused all traffic to pile on one node while others idled. Changes: - Collect all deployments tied at the minimum count into a list and randomly pick from them, so load is spread evenly when counts are equal - Skip stale cache entries (deployments no longer in the healthy list) when scanning the count dict, preventing phantom entries from influencing routing decisions - Add unit tests covering equal-count distribution, unique-minimum selection, stale-entry filtering, and sync/async cache paths Made-with: Cursor --- litellm/router_strategy/least_busy.py | 32 +++-- .../router_strategy/test_least_busy.py | 124 ++++++++++++++++++ 2 files changed, 142 insertions(+), 14 deletions(-) create mode 100644 tests/test_litellm/router_strategy/test_least_busy.py diff --git a/litellm/router_strategy/least_busy.py b/litellm/router_strategy/least_busy.py index e1614388379..74b142a1d8b 100644 --- a/litellm/router_strategy/least_busy.py +++ b/litellm/router_strategy/least_busy.py @@ -197,27 +197,31 @@ class LeastBusyLoggingHandler(CustomLogger): """ Helper to get deployments using least busy strategy """ + healthy_ids = set() for d in healthy_deployments: - ## if healthy deployment not yet used - if d["model_info"]["id"] not in all_deployments: - all_deployments[d["model_info"]["id"]] = 0 - # map deployment to id - # pick least busy deployment + _id = d["model_info"]["id"] + healthy_ids.add(_id) + if _id not in all_deployments: + all_deployments[_id] = 0 + min_traffic = float("inf") - min_deployment = None + min_deployment_ids: list = [] for k, v in all_deployments.items(): + if k not in healthy_ids: + continue if v < min_traffic: min_traffic = v - min_deployment = k - if min_deployment is not None: - ## check if min deployment is a string, if so, cast it to int + min_deployment_ids = [k] + elif v == min_traffic: + min_deployment_ids.append(k) + + if min_deployment_ids: + chosen_id = random.choice(min_deployment_ids) for m in healthy_deployments: - if m["model_info"]["id"] == min_deployment: + if m["model_info"]["id"] == chosen_id: return m - min_deployment = random.choice(healthy_deployments) - else: - min_deployment = random.choice(healthy_deployments) - return min_deployment + + return random.choice(healthy_deployments) def get_available_deployments( self, diff --git a/tests/test_litellm/router_strategy/test_least_busy.py b/tests/test_litellm/router_strategy/test_least_busy.py new file mode 100644 index 00000000000..a1d678e35c9 --- /dev/null +++ b/tests/test_litellm/router_strategy/test_least_busy.py @@ -0,0 +1,124 @@ +import os +import sys +from collections import Counter + +import pytest + +sys.path.insert( + 0, os.path.abspath("../../..") +) # Adds the parent directory to the system path + +from litellm.caching.caching import DualCache +from litellm.router_strategy.least_busy import LeastBusyLoggingHandler + + +def _make_deployment(id_val): + return { + "model_name": "test-model", + "litellm_params": {"model": "openai/gpt-4.1-mini"}, + "model_info": {"id": str(id_val)}, + } + + +class TestLeastBusyTieBreaking: + """Tests that least-busy strategy distributes requests across tied deployments.""" + + def test_should_randomly_distribute_when_all_counts_are_zero(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + deployments = [_make_deployment(i) for i in range(3)] + + selections = Counter() + for _ in range(300): + chosen = handler._get_available_deployments( + healthy_deployments=deployments, all_deployments={} + ) + selections[chosen["model_info"]["id"]] += 1 + + assert len(selections) == 3, ( + f"Expected all 3 deployments to be selected at least once, got: {selections}" + ) + for dep_id, count in selections.items(): + assert count > 30, ( + f"Deployment {dep_id} selected only {count}/300 times — distribution is too skewed" + ) + + def test_should_randomly_distribute_when_counts_are_tied(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + deployments = [_make_deployment(i) for i in range(2)] + tied_counts = {"0": 5, "1": 5} + + selections = Counter() + for _ in range(200): + chosen = handler._get_available_deployments( + healthy_deployments=deployments, + all_deployments=dict(tied_counts), + ) + selections[chosen["model_info"]["id"]] += 1 + + assert len(selections) == 2, ( + f"Expected both deployments to be selected, got: {selections}" + ) + for dep_id, count in selections.items(): + assert count > 40, ( + f"Deployment {dep_id} selected only {count}/200 times — distribution is too skewed" + ) + + def test_should_pick_unique_minimum(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + deployments = [_make_deployment(i) for i in range(3)] + counts = {"0": 10, "1": 1, "2": 50} + + for _ in range(50): + chosen = handler._get_available_deployments( + healthy_deployments=deployments, + all_deployments=dict(counts), + ) + assert chosen["model_info"]["id"] == "1" + + def test_should_skip_unhealthy_deployments_in_cache(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + healthy = [_make_deployment(1), _make_deployment(2)] + counts_with_stale = {"0": 0, "1": 5, "2": 10} + + for _ in range(50): + chosen = handler._get_available_deployments( + healthy_deployments=healthy, + all_deployments=dict(counts_with_stale), + ) + assert chosen["model_info"]["id"] in ("1", "2") + assert chosen["model_info"]["id"] == "1" + + +class TestLeastBusyGetAvailableDeployments: + """Tests the sync/async wrappers for cache interaction.""" + + def test_should_use_cache_for_selection(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + deployments = [_make_deployment(0), _make_deployment(1)] + + cache.set_cache( + key="test-model_request_count", value={"0": 10, "1": 2} + ) + chosen = handler.get_available_deployments( + model_group="test-model", healthy_deployments=deployments + ) + assert chosen["model_info"]["id"] == "1" + + @pytest.mark.asyncio + async def test_should_use_cache_for_async_selection(self): + cache = DualCache() + handler = LeastBusyLoggingHandler(router_cache=cache) + deployments = [_make_deployment(0), _make_deployment(1)] + + await cache.async_set_cache( + key="test-model_request_count", value={"0": 10, "1": 2} + ) + chosen = await handler.async_get_available_deployments( + model_group="test-model", healthy_deployments=deployments + ) + assert chosen["model_info"]["id"] == "1"