From 90cef7c7741c4c23fab0474018590287186cf255 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Tue, 12 Aug 2025 23:32:23 -0700 Subject: [PATCH] fix(router.py): fix cooldown increment logic --- litellm/router.py | 31 ++++++------ tests/local_testing/test_router_cooldowns.py | 51 +++++++++++++------- 2 files changed, 49 insertions(+), 33 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index b5dac3263c4..3fee34fa5c0 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4351,16 +4351,22 @@ class Router: tpm_model_info = deployment_model_info.get("tpm", None) rpm_model_info = deployment_model_info.get("rpm", None) - ## if all are none, return - no need to track current tpm/rpm usage for models with no tpm/rpm set - if ( - tpm is None - and rpm is None - and tpm_litellm_params is None - and rpm_litellm_params is None - and tpm_model_info is None - and rpm_model_info is None - ): - return + # Always track deployment successes for cooldown logic, regardless of TPM/RPM limits + increment_deployment_successes_for_current_minute( + litellm_router_instance=self, + deployment_id=id, + ) + + ## if all are none, return - no need to track current tpm/rpm usage for models with no tpm/rpm set + if ( + tpm is None + and rpm is None + and tpm_litellm_params is None + and rpm_litellm_params is None + and tpm_model_info is None + and rpm_model_info is None + ): + return parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs) total_tokens: float = standard_logging_object.get("total_tokens", 0) @@ -4409,11 +4415,6 @@ class Router: parent_otel_span=parent_otel_span, ) - increment_deployment_successes_for_current_minute( - litellm_router_instance=self, - deployment_id=id, - ) - return tpm_key except Exception as e: diff --git a/tests/local_testing/test_router_cooldowns.py b/tests/local_testing/test_router_cooldowns.py index 2a04bcc89a8..cd178e2aaee 100644 --- a/tests/local_testing/test_router_cooldowns.py +++ b/tests/local_testing/test_router_cooldowns.py @@ -22,7 +22,10 @@ import openai import litellm from litellm import Router from litellm.integrations.custom_logger import CustomLogger -from litellm.router_utils.cooldown_handlers import _async_get_cooldown_deployments, _should_run_cooldown_logic +from litellm.router_utils.cooldown_handlers import ( + _async_get_cooldown_deployments, + _should_run_cooldown_logic, +) from litellm.types.router import ( DeploymentTypedDict, LiteLLMParamsTypedDict, @@ -148,7 +151,9 @@ async def test_cooldown_time_zero_uses_zero_not_default(): ) # Mock the add_deployment_to_cooldown method to verify it's NOT called - with patch.object(router.cooldown_cache, "add_deployment_to_cooldown") as mock_add_cooldown: + with patch.object( + router.cooldown_cache, "add_deployment_to_cooldown" + ) as mock_add_cooldown: try: await router.acompletion( model="gpt-3.5-turbo", @@ -160,13 +165,13 @@ async def test_cooldown_time_zero_uses_zero_not_default(): # Verify that add_deployment_to_cooldown was NOT called due to early exit mock_add_cooldown.assert_not_called() - + # Also verify the deployment is not in cooldown cooldown_list = await _async_get_cooldown_deployments( litellm_router_instance=router, parent_otel_span=None ) assert len(cooldown_list) == 0 - + # Verify the deployment is still healthy and available healthy_deployments, _ = await router._async_get_healthy_deployments( model="gpt-3.5-turbo", parent_otel_span=None @@ -192,44 +197,54 @@ def test_should_run_cooldown_logic_early_exit_on_zero_cooldown(): ], cooldown_time=300, ) - + # Test with time_to_cooldown = 0 - should return False (don't run cooldown logic) result = _should_run_cooldown_logic( litellm_router_instance=router, deployment="test-deployment-id", exception_status=429, - original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"), - time_to_cooldown=0.0 + original_exception=litellm.RateLimitError( + "test error", "openai", "gpt-3.5-turbo" + ), + time_to_cooldown=0.0, ) assert result is False, "Should not run cooldown logic when time_to_cooldown is 0" - + # Test with very small time_to_cooldown (effectively 0) - should return False result = _should_run_cooldown_logic( litellm_router_instance=router, deployment="test-deployment-id", exception_status=429, - original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"), - time_to_cooldown=1e-10 + original_exception=litellm.RateLimitError( + "test error", "openai", "gpt-3.5-turbo" + ), + time_to_cooldown=1e-10, ) - assert result is False, "Should not run cooldown logic when time_to_cooldown is effectively 0" - + assert ( + result is False + ), "Should not run cooldown logic when time_to_cooldown is effectively 0" + # Test with None time_to_cooldown - should return True (use default cooldown logic) result = _should_run_cooldown_logic( litellm_router_instance=router, - deployment="test-deployment-id", + deployment="test-deployment-id", exception_status=429, - original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"), - time_to_cooldown=None + original_exception=litellm.RateLimitError( + "test error", "openai", "gpt-3.5-turbo" + ), + time_to_cooldown=None, ) assert result is True, "Should run cooldown logic when time_to_cooldown is None" - + # Test with positive time_to_cooldown - should return True result = _should_run_cooldown_logic( litellm_router_instance=router, deployment="test-deployment-id", exception_status=429, - original_exception=litellm.RateLimitError("test error", "openai", "gpt-3.5-turbo"), - time_to_cooldown=60.0 + original_exception=litellm.RateLimitError( + "test error", "openai", "gpt-3.5-turbo" + ), + time_to_cooldown=60.0, ) assert result is True, "Should run cooldown logic when time_to_cooldown is positive"