From 9bbb4491d0cae71398c085b0b33be095de706965 Mon Sep 17 00:00:00 2001 From: Sparrow He Date: Sat, 18 Apr 2026 21:42:56 +0800 Subject: [PATCH] fix(router): randomize deployment selection on exact cost ties in lowest_cost strategy Previously, the `lowest_cost` routing strategy used Python's stable `sorted()` on costs alone, causing it to consistently pick the exact same deployment when multiple deployments had identical costs. The original code comment mentioned "randomly sample", but the implementation lacked randomness. - Updated `lowest_cost.py` to sort by `(cost, random.random())` to break ties randomly, achieving true load balancing across equally priced deployments. - Added `test_lowest_cost_routing_randomization` to `test_lowest_cost_routing.py` to ensure traffic is randomly distributed when deployments have matching `input_cost_per_token` and `output_cost_per_token`. --- litellm/router_strategy/lowest_cost.py | 7 ++- .../local_testing/test_lowest_cost_routing.py | 50 +++++++++++++++++++ 2 files changed, 55 insertions(+), 2 deletions(-) diff --git a/litellm/router_strategy/lowest_cost.py b/litellm/router_strategy/lowest_cost.py index 54498363f51..0f1149d7e1e 100644 --- a/litellm/router_strategy/lowest_cost.py +++ b/litellm/router_strategy/lowest_cost.py @@ -1,6 +1,7 @@ #### What this does #### # picks based on response time (for streaming, this is time to first token) from datetime import datetime, timedelta +import random from typing import Dict, List, Optional, Union import litellm @@ -232,12 +233,14 @@ class LowestCostLoggingHandler(CustomLogger): input_tokens = 0 # randomly sample from all_deployments, incase all deployments have latency=0.0 - _items = all_deployments.items() + _items = random.sample( + list(all_deployments.items()), len(all_deployments.items()) + ) ### GET AVAILABLE DEPLOYMENTS ### filter out any deployments > tpm/rpm limits potential_deployments = [] _cost_per_deployment = {} - for item, item_map in all_deployments.items(): + for item, item_map in _items: ## get the item from model list _deployment = None for m in healthy_deployments: diff --git a/tests/local_testing/test_lowest_cost_routing.py b/tests/local_testing/test_lowest_cost_routing.py index 4e8b06fb628..b39d8b107c9 100644 --- a/tests/local_testing/test_lowest_cost_routing.py +++ b/tests/local_testing/test_lowest_cost_routing.py @@ -202,3 +202,53 @@ async def test_get_available_endpoints_tpm_rpm_check_async(ans_rpm): assert (d_ans and d_ans["model_info"]["id"]) == ans print("selected deployment:", d_ans) + + +@pytest.mark.asyncio +async def test_lowest_cost_routing_randomization(): + """ + Test that when models have the exact same cost, the router randomizes the selection. + """ + test_cache = DualCache() + model_list = [ + { + "model_name": "gpt-3.5-turbo", + "litellm_params": { + "model": "azure/gpt-3.5-turbo-1", + "input_cost_per_token": 0.001, + "output_cost_per_token": 0.002, + }, + "model_info": {"id": "model-1"}, + }, + { + "model_name": "gpt-3.5-turbo", + "litellm_params": { + "model": "azure/gpt-3.5-turbo-2", + "input_cost_per_token": 0.001, + "output_cost_per_token": 0.002, + }, + "model_info": {"id": "model-2"}, + }, + ] + + lowest_cost_logger = LowestCostLoggingHandler( + router_cache=test_cache, + ) + model_group = "gpt-3.5-turbo" + + selected_counts = {"model-1": 0, "model-2": 0} + + # Run multiple times to observe randomization + for _ in range(50): + selected_model = await lowest_cost_logger.async_get_available_deployments( + model_group=model_group, healthy_deployments=model_list + ) + selected_id = selected_model["model_info"]["id"] + selected_counts[selected_id] += 1 + + print("selected counts:", selected_counts) + + # Both models should roughly be selected since they have the exact same cost. + # To avoid flakiness, just check that > 0 for both (or at least > 5 in 50 tries). + assert selected_counts["model-1"] > 0 + assert selected_counts["model-2"] > 0