fix(router.py): fix cooldown increment logic

This commit is contained in:
Krrish Dholakia 2025-08-12 23:32:23 -07:00
parent d04eb62539
commit 90cef7c774
2 changed files with 49 additions and 33 deletions

View file

@ -4351,16 +4351,22 @@ class Router:
tpm_model_info = deployment_model_info.get("tpm", None)
rpm_model_info = deployment_model_info.get("rpm", None)
## if all are none, return - no need to track current tpm/rpm usage for models with no tpm/rpm set
if (
tpm is None
and rpm is None
and tpm_litellm_params is None
and rpm_litellm_params is None
and tpm_model_info is None
and rpm_model_info is None
):
return
# Always track deployment successes for cooldown logic, regardless of TPM/RPM limits
increment_deployment_successes_for_current_minute(
litellm_router_instance=self,
deployment_id=id,
)
## if all are none, return - no need to track current tpm/rpm usage for models with no tpm/rpm set
if (
tpm is None
and rpm is None
and tpm_litellm_params is None
and rpm_litellm_params is None
and tpm_model_info is None
and rpm_model_info is None
):
return
parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs)
total_tokens: float = standard_logging_object.get("total_tokens", 0)
@ -4409,11 +4415,6 @@ class Router:
parent_otel_span=parent_otel_span,
)
increment_deployment_successes_for_current_minute(
litellm_router_instance=self,
deployment_id=id,
)
return tpm_key
except Exception as e:

View file

@ -22,7 +22,10 @@ import openai
import litellm
from litellm import Router
from litellm.integrations.custom_logger import CustomLogger
from litellm.router_utils.cooldown_handlers import _async_get_cooldown_deployments, _should_run_cooldown_logic
from litellm.router_utils.cooldown_handlers import (
_async_get_cooldown_deployments,
_should_run_cooldown_logic,
)
from litellm.types.router import (
DeploymentTypedDict,
LiteLLMParamsTypedDict,
@ -148,7 +151,9 @@ async def test_cooldown_time_zero_uses_zero_not_default():
)
# Mock the add_deployment_to_cooldown method to verify it's NOT called
with patch.object(router.cooldown_cache, "add_deployment_to_cooldown") as mock_add_cooldown:
with patch.object(
router.cooldown_cache, "add_deployment_to_cooldown"
) as mock_add_cooldown:
try:
await router.acompletion(
model="gpt-3.5-turbo",
@ -160,13 +165,13 @@ async def test_cooldown_time_zero_uses_zero_not_default():
# Verify that add_deployment_to_cooldown was NOT called due to early exit
mock_add_cooldown.assert_not_called()
# Also verify the deployment is not in cooldown
cooldown_list = await _async_get_cooldown_deployments(
litellm_router_instance=router, parent_otel_span=None
)
assert len(cooldown_list) == 0
# Verify the deployment is still healthy and available
healthy_deployments, _ = await router._async_get_healthy_deployments(
model="gpt-3.5-turbo", parent_otel_span=None
@ -192,44 +197,54 @@ def test_should_run_cooldown_logic_early_exit_on_zero_cooldown():
],
cooldown_time=300,
)
# Test with time_to_cooldown = 0 - should return False (don't run cooldown logic)
result = _should_run_cooldown_logic(
litellm_router_instance=router,
deployment="test-deployment-id",
exception_status=429,
original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"),
time_to_cooldown=0.0
original_exception=litellm.RateLimitError(
"test error", "openai", "gpt-3.5-turbo"
),
time_to_cooldown=0.0,
)
assert result is False, "Should not run cooldown logic when time_to_cooldown is 0"
# Test with very small time_to_cooldown (effectively 0) - should return False
result = _should_run_cooldown_logic(
litellm_router_instance=router,
deployment="test-deployment-id",
exception_status=429,
original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"),
time_to_cooldown=1e-10
original_exception=litellm.RateLimitError(
"test error", "openai", "gpt-3.5-turbo"
),
time_to_cooldown=1e-10,
)
assert result is False, "Should not run cooldown logic when time_to_cooldown is effectively 0"
assert (
result is False
), "Should not run cooldown logic when time_to_cooldown is effectively 0"
# Test with None time_to_cooldown - should return True (use default cooldown logic)
result = _should_run_cooldown_logic(
litellm_router_instance=router,
deployment="test-deployment-id",
deployment="test-deployment-id",
exception_status=429,
original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"),
time_to_cooldown=None
original_exception=litellm.RateLimitError(
"test error", "openai", "gpt-3.5-turbo"
),
time_to_cooldown=None,
)
assert result is True, "Should run cooldown logic when time_to_cooldown is None"
# Test with positive time_to_cooldown - should return True
result = _should_run_cooldown_logic(
litellm_router_instance=router,
deployment="test-deployment-id",
exception_status=429,
original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"),
time_to_cooldown=60.0
original_exception=litellm.RateLimitError(
"test error", "openai", "gpt-3.5-turbo"
),
time_to_cooldown=60.0,
)
assert result is True, "Should run cooldown logic when time_to_cooldown is positive"