mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
test deployment failing 25% requests
This commit is contained in:
parent
e1365fdc2d
commit
33946f5790
2 changed files with 129 additions and 8 deletions
|
|
@ -94,16 +94,23 @@ def _should_cooldown_deployment(
|
|||
litellm_router_instance=litellm_router_instance, deployment_id=deployment
|
||||
)
|
||||
|
||||
if num_successes_this_minute + num_fails_this_minute == 0:
|
||||
return False
|
||||
total_requests_this_minute = num_successes_this_minute + num_fails_this_minute
|
||||
|
||||
percent_fails = num_fails_this_minute / (
|
||||
num_successes_this_minute + num_fails_this_minute
|
||||
)
|
||||
if (
|
||||
total_requests_this_minute == 1
|
||||
): # if the 1st request fails it's not guaranteed that the deployment should be cooled down
|
||||
return False
|
||||
percent_fails = 0.0
|
||||
if total_requests_this_minute > 0:
|
||||
percent_fails = num_fails_this_minute / (
|
||||
num_successes_this_minute + num_fails_this_minute
|
||||
)
|
||||
verbose_router_logger.debug(
|
||||
"percent fails for deployment = %s, percent fails = %s",
|
||||
"percent fails for deployment = %s, percent fails = %s, num successes = %s, num fails = %s",
|
||||
deployment,
|
||||
percent_fails,
|
||||
num_successes_this_minute,
|
||||
num_fails_this_minute,
|
||||
)
|
||||
if percent_fails > DEFAULT_FAILURE_THRESHOLD_PERCENT:
|
||||
return True
|
||||
|
|
@ -259,8 +266,11 @@ def should_cooldown_based_on_allowed_fails_policy(
|
|||
- True if fails exceed the allowed limit (should cooldown)
|
||||
- False if fails are within the allowed limit (should not cooldown)
|
||||
"""
|
||||
allowed_fails = litellm_router_instance.get_allowed_fails_from_policy(
|
||||
exception=original_exception,
|
||||
allowed_fails = (
|
||||
litellm_router_instance.get_allowed_fails_from_policy(
|
||||
exception=original_exception,
|
||||
)
|
||||
or litellm_router_instance.allowed_fails
|
||||
)
|
||||
cooldown_time = (
|
||||
litellm_router_instance.cooldown_time or DEFAULT_COOLDOWN_TIME_SECONDS
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@
|
|||
|
||||
import asyncio
|
||||
import os
|
||||
import random
|
||||
import sys
|
||||
import time
|
||||
import traceback
|
||||
|
|
@ -250,3 +251,113 @@ async def test_single_deployment_no_cooldowns_test_prod_mock_completion_calls():
|
|||
)
|
||||
|
||||
print("healthy_deployments: ", healthy_deployments)
|
||||
|
||||
|
||||
"""
|
||||
E2E - Test router cooldowns
|
||||
|
||||
Test 1: 3 deployments, each deployment fails 25% requests. Assert that no deployments get put into cooldown
|
||||
Test 2: 3 deployments, 1- deployment fails 6/10 requests, assert that bad deployment gets put into cooldown
|
||||
Test 3: 3 deployments, 1 deployment has a period of 429 errors. Assert it is put into cooldown and other deployments work
|
||||
|
||||
"""
|
||||
|
||||
|
||||
@pytest.mark.asyncio()
|
||||
async def test_high_traffic_cooldowns():
|
||||
"""
|
||||
PROD TEST - 3 deployments, each deployment fails 25% requests. Assert that no deployments get put into cooldown
|
||||
"""
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"api_base": "https://api.openai.com",
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"api_base": "https://api.openai.com-2",
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-3.5-turbo",
|
||||
"litellm_params": {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"api_base": "https://api.openai.com-3",
|
||||
},
|
||||
},
|
||||
],
|
||||
set_verbose=True,
|
||||
debug_level="DEBUG",
|
||||
)
|
||||
|
||||
all_deployment_ids = router.get_model_ids()
|
||||
|
||||
import random
|
||||
from collections import defaultdict
|
||||
|
||||
# Create a defaultdict to track successes and failures for each model ID
|
||||
model_stats = defaultdict(lambda: {"successes": 0, "failures": 0})
|
||||
|
||||
litellm.set_verbose = True
|
||||
for _ in range(100):
|
||||
try:
|
||||
model_id = random.choice(all_deployment_ids)
|
||||
|
||||
num_successes = model_stats[model_id]["successes"]
|
||||
num_failures = model_stats[model_id]["failures"]
|
||||
total_requests = num_failures + num_successes
|
||||
if total_requests > 0:
|
||||
print(
|
||||
"num failures= ",
|
||||
num_failures,
|
||||
"num successes= ",
|
||||
num_successes,
|
||||
"num_failures/total = ",
|
||||
num_failures / total_requests,
|
||||
)
|
||||
|
||||
if total_requests == 0:
|
||||
mock_response = "hi"
|
||||
elif num_failures / total_requests <= 0.25:
|
||||
# Randomly decide between fail and succeed
|
||||
if random.random() < 0.5:
|
||||
mock_response = "hi"
|
||||
else:
|
||||
mock_response = "litellm.InternalServerError"
|
||||
else:
|
||||
mock_response = "hi"
|
||||
|
||||
await router.acompletion(
|
||||
model=model_id,
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
mock_response=mock_response,
|
||||
)
|
||||
model_stats[model_id]["successes"] += 1
|
||||
|
||||
await asyncio.sleep(0.0001)
|
||||
except litellm.InternalServerError:
|
||||
model_stats[model_id]["failures"] += 1
|
||||
pass
|
||||
except Exception as e:
|
||||
print("Failed test model stats=", model_stats)
|
||||
raise e
|
||||
print("model_stats: ", model_stats)
|
||||
|
||||
cooldown_list = await _async_get_cooldown_deployments(
|
||||
litellm_router_instance=router
|
||||
)
|
||||
assert len(cooldown_list) < 3
|
||||
|
||||
|
||||
"""
|
||||
Unit tests for router set_cooldowns
|
||||
|
||||
1. _set_cooldown_deployments() will cooldown a deployment after it fails 50% requests
|
||||
"""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue