fix(router): randomize deployment selection on exact cost ties in lowest_cost strategy

Previously, the `lowest_cost` routing strategy used Python's stable `sorted()` on costs alone, causing it to consistently pick the exact same deployment when multiple deployments had identical costs. The original code comment mentioned "randomly sample", but the implementation lacked randomness.

- Updated `lowest_cost.py` to sort by `(cost, random.random())` to break ties randomly, achieving true load balancing across equally priced deployments.
- Added `test_lowest_cost_routing_randomization` to `test_lowest_cost_routing.py` to ensure traffic is randomly distributed when deployments have matching `input_cost_per_token` and `output_cost_per_token`.
This commit is contained in:
Sparrow He 2026-04-18 21:42:56 +08:00
parent 0b50a29baf
commit 9bbb4491d0
No known key found for this signature in database
GPG key ID: E7449770C1B09D2B
2 changed files with 55 additions and 2 deletions

View file

@ -1,6 +1,7 @@
#### What this does ####
# picks based on response time (for streaming, this is time to first token)
from datetime import datetime, timedelta
import random
from typing import Dict, List, Optional, Union
import litellm
@ -232,12 +233,14 @@ class LowestCostLoggingHandler(CustomLogger):
input_tokens = 0
# randomly sample from all_deployments, incase all deployments have latency=0.0
_items = all_deployments.items()
_items = random.sample(
list(all_deployments.items()), len(all_deployments.items())
)
### GET AVAILABLE DEPLOYMENTS ### filter out any deployments > tpm/rpm limits
potential_deployments = []
_cost_per_deployment = {}
for item, item_map in all_deployments.items():
for item, item_map in _items:
## get the item from model list
_deployment = None
for m in healthy_deployments:

View file

@ -202,3 +202,53 @@ async def test_get_available_endpoints_tpm_rpm_check_async(ans_rpm):
assert (d_ans and d_ans["model_info"]["id"]) == ans
print("selected deployment:", d_ans)
@pytest.mark.asyncio
async def test_lowest_cost_routing_randomization():
"""
Test that when models have the exact same cost, the router randomizes the selection.
"""
test_cache = DualCache()
model_list = [
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {
"model": "azure/gpt-3.5-turbo-1",
"input_cost_per_token": 0.001,
"output_cost_per_token": 0.002,
},
"model_info": {"id": "model-1"},
},
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {
"model": "azure/gpt-3.5-turbo-2",
"input_cost_per_token": 0.001,
"output_cost_per_token": 0.002,
},
"model_info": {"id": "model-2"},
},
]
lowest_cost_logger = LowestCostLoggingHandler(
router_cache=test_cache,
)
model_group = "gpt-3.5-turbo"
selected_counts = {"model-1": 0, "model-2": 0}
# Run multiple times to observe randomization
for _ in range(50):
selected_model = await lowest_cost_logger.async_get_available_deployments(
model_group=model_group, healthy_deployments=model_list
)
selected_id = selected_model["model_info"]["id"]
selected_counts[selected_id] += 1
print("selected counts:", selected_counts)
# Both models should roughly be selected since they have the exact same cost.
# To avoid flakiness, just check that > 0 for both (or at least > 5 in 50 tries).
assert selected_counts["model-1"] > 0
assert selected_counts["model-2"] > 0