mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(router): randomize deployment selection on exact cost ties in lowest_cost strategy
Previously, the `lowest_cost` routing strategy used Python's stable `sorted()` on costs alone, causing it to consistently pick the exact same deployment when multiple deployments had identical costs. The original code comment mentioned "randomly sample", but the implementation lacked randomness. - Updated `lowest_cost.py` to sort by `(cost, random.random())` to break ties randomly, achieving true load balancing across equally priced deployments. - Added `test_lowest_cost_routing_randomization` to `test_lowest_cost_routing.py` to ensure traffic is randomly distributed when deployments have matching `input_cost_per_token` and `output_cost_per_token`.
This commit is contained in:
parent
0b50a29baf
commit
9bbb4491d0
2 changed files with 55 additions and 2 deletions
|
|
@ -1,6 +1,7 @@
|
|||
#### What this does ####
|
||||
# picks based on response time (for streaming, this is time to first token)
|
||||
from datetime import datetime, timedelta
|
||||
import random
|
||||
from typing import Dict, List, Optional, Union
|
||||
|
||||
import litellm
|
||||
|
|
@ -232,12 +233,14 @@ class LowestCostLoggingHandler(CustomLogger):
|
|||
input_tokens = 0
|
||||
|
||||
# randomly sample from all_deployments, incase all deployments have latency=0.0
|
||||
_items = all_deployments.items()
|
||||
_items = random.sample(
|
||||
list(all_deployments.items()), len(all_deployments.items())
|
||||
)
|
||||
|
||||
### GET AVAILABLE DEPLOYMENTS ### filter out any deployments > tpm/rpm limits
|
||||
potential_deployments = []
|
||||
_cost_per_deployment = {}
|
||||
for item, item_map in all_deployments.items():
|
||||
for item, item_map in _items:
|
||||
## get the item from model list
|
||||
_deployment = None
|
||||
for m in healthy_deployments:
|
||||
|
|
|
|||
|
|
@ -202,3 +202,53 @@ async def test_get_available_endpoints_tpm_rpm_check_async(ans_rpm):
|
|||
assert (d_ans and d_ans["model_info"]["id"]) == ans
|
||||
|
||||
print("selected deployment:", d_ans)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_lowest_cost_routing_randomization():
|
||||
"""
|
||||
Test that when models have the exact same cost, the router randomizes the selection.
|
||||
"""
|
||||
test_cache = DualCache()
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-3.5-turbo-1",
|
||||
"input_cost_per_token": 0.001,
|
||||
"output_cost_per_token": 0.002,
|
||||
},
|
||||
"model_info": {"id": "model-1"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-3.5-turbo-2",
|
||||
"input_cost_per_token": 0.001,
|
||||
"output_cost_per_token": 0.002,
|
||||
},
|
||||
"model_info": {"id": "model-2"},
|
||||
},
|
||||
]
|
||||
|
||||
lowest_cost_logger = LowestCostLoggingHandler(
|
||||
router_cache=test_cache,
|
||||
)
|
||||
model_group = "gpt-3.5-turbo"
|
||||
|
||||
selected_counts = {"model-1": 0, "model-2": 0}
|
||||
|
||||
# Run multiple times to observe randomization
|
||||
for _ in range(50):
|
||||
selected_model = await lowest_cost_logger.async_get_available_deployments(
|
||||
model_group=model_group, healthy_deployments=model_list
|
||||
)
|
||||
selected_id = selected_model["model_info"]["id"]
|
||||
selected_counts[selected_id] += 1
|
||||
|
||||
print("selected counts:", selected_counts)
|
||||
|
||||
# Both models should roughly be selected since they have the exact same cost.
|
||||
# To avoid flakiness, just check that > 0 for both (or at least > 5 in 50 tries).
|
||||
assert selected_counts["model-1"] > 0
|
||||
assert selected_counts["model-2"] > 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue