mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-21 00:21:49 +00:00
feat(router): add health_check_ignore_transient_errors flag
When enabled, health check failures with 429 (rate limit) or 408 (timeout) status codes are skipped from the cooldown pipeline. These are transient load issues, not broken deployments. Auth errors (401), 404, and 5xx errors still increment counters and trigger cooldown as before. Config (general_settings): health_check_ignore_transient_errors: true Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
1867ca51ae
commit
b8391bc17c
3 changed files with 147 additions and 0 deletions
|
|
@ -2147,6 +2147,12 @@ def _write_health_state_to_router_cache(
|
|||
|
||||
exception_status = getattr(original_exception, "status_code", 500)
|
||||
|
||||
if llm_router.health_check_ignore_transient_errors and exception_status in (
|
||||
429,
|
||||
408,
|
||||
):
|
||||
continue
|
||||
|
||||
increment_deployment_failures_for_current_minute(
|
||||
litellm_router_instance=llm_router,
|
||||
deployment_id=model_id,
|
||||
|
|
@ -3107,6 +3113,7 @@ class ProxyConfig:
|
|||
general_settings = {}
|
||||
_enable_hc_routing = False
|
||||
_hc_staleness = None
|
||||
_hc_ignore_transient = False
|
||||
if general_settings:
|
||||
### LOAD KEY MANAGEMENT SETTINGS FIRST (needed for custom secret manager) ###
|
||||
key_management_settings = general_settings.get(
|
||||
|
|
@ -3293,6 +3300,9 @@ class ProxyConfig:
|
|||
_hc_staleness = general_settings.get(
|
||||
"health_check_staleness_threshold", None
|
||||
)
|
||||
_hc_ignore_transient = general_settings.get(
|
||||
"health_check_ignore_transient_errors", False
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
"background_health_check_config enabled=%s shared=%s interval_seconds=%s max_concurrency=%s details=%s health_check_routing=%s",
|
||||
use_background_health_checks,
|
||||
|
|
@ -3335,6 +3345,8 @@ class ProxyConfig:
|
|||
router_params["enable_health_check_routing"] = True
|
||||
if _hc_staleness is not None:
|
||||
router_params["health_check_staleness_threshold"] = _hc_staleness
|
||||
if _hc_ignore_transient:
|
||||
router_params["health_check_ignore_transient_errors"] = True
|
||||
## MODEL LIST
|
||||
model_list = config.get("model_list", None)
|
||||
if model_list:
|
||||
|
|
|
|||
|
|
@ -306,6 +306,7 @@ class Router:
|
|||
ignore_invalid_deployments: bool = False,
|
||||
enable_health_check_routing: bool = False,
|
||||
health_check_staleness_threshold: Optional[int] = None,
|
||||
health_check_ignore_transient_errors: bool = False,
|
||||
) -> None:
|
||||
"""
|
||||
Initialize the Router class with the given parameters for caching, reliability, and routing strategy.
|
||||
|
|
@ -495,6 +496,7 @@ class Router:
|
|||
)
|
||||
self.disable_cooldowns = disable_cooldowns
|
||||
self.enable_health_check_routing = enable_health_check_routing
|
||||
self.health_check_ignore_transient_errors = health_check_ignore_transient_errors
|
||||
_staleness = health_check_staleness_threshold or (
|
||||
DEFAULT_HEALTH_CHECK_INTERVAL * DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER
|
||||
)
|
||||
|
|
|
|||
|
|
@ -530,3 +530,136 @@ class TestAllDeploymentsInCooldownSafetyNet:
|
|||
assert (
|
||||
len(filtered) == 2
|
||||
), "Safety net should return all deployments when all are in cooldown"
|
||||
|
||||
|
||||
class TestHealthCheckIgnoreTransientErrors:
|
||||
"""
|
||||
When health_check_ignore_transient_errors=True, health check failures with
|
||||
429 or 408 status codes should NOT increment failure counters or trigger cooldown.
|
||||
401, 404, and 5xx errors should still be processed normally.
|
||||
"""
|
||||
|
||||
def test_429_skipped_when_flag_enabled(self):
|
||||
"""429 from health check does not trigger cooldown when flag is set."""
|
||||
import litellm.proxy.proxy_server as proxy_module
|
||||
from litellm.proxy.proxy_server import _write_health_state_to_router_cache
|
||||
|
||||
router = Router(
|
||||
model_list=[_make_model("deploy-1"), _make_model("deploy-2", "gpt-5")],
|
||||
allowed_fails_policy=AllowedFailsPolicy(RateLimitErrorAllowedFails=0),
|
||||
enable_health_check_routing=True,
|
||||
health_check_ignore_transient_errors=True,
|
||||
)
|
||||
|
||||
rate_exc = litellm.RateLimitError(
|
||||
message="Rate limited", model="gpt-4", llm_provider="openai"
|
||||
)
|
||||
assert getattr(rate_exc, "status_code", None) == 429
|
||||
|
||||
unhealthy_endpoints = [
|
||||
{"model_id": "deploy-1", "error": "rate limited", "exception": rate_exc},
|
||||
]
|
||||
|
||||
with patch.object(proxy_module, "llm_router", router):
|
||||
with patch(
|
||||
"litellm.router_utils.cooldown_handlers._set_cooldown_deployments"
|
||||
) as mock_cooldown:
|
||||
with patch(
|
||||
"litellm.router_utils.router_callbacks.track_deployment_metrics.increment_deployment_failures_for_current_minute"
|
||||
) as mock_increment:
|
||||
_write_health_state_to_router_cache(
|
||||
healthy_endpoints=[],
|
||||
unhealthy_endpoints=unhealthy_endpoints,
|
||||
)
|
||||
mock_cooldown.assert_not_called()
|
||||
mock_increment.assert_not_called()
|
||||
|
||||
def test_408_skipped_when_flag_enabled(self):
|
||||
"""408 from health check does not trigger cooldown when flag is set."""
|
||||
import litellm.proxy.proxy_server as proxy_module
|
||||
from litellm.proxy.proxy_server import _write_health_state_to_router_cache
|
||||
|
||||
router = Router(
|
||||
model_list=[_make_model("deploy-1"), _make_model("deploy-2", "gpt-5")],
|
||||
allowed_fails_policy=AllowedFailsPolicy(TimeoutErrorAllowedFails=0),
|
||||
enable_health_check_routing=True,
|
||||
health_check_ignore_transient_errors=True,
|
||||
)
|
||||
|
||||
timeout_exc = litellm.Timeout(
|
||||
message="Health check timeout exceeded", model="", llm_provider=""
|
||||
)
|
||||
|
||||
unhealthy_endpoints = [
|
||||
{"model_id": "deploy-1", "error": "timeout", "exception": timeout_exc},
|
||||
]
|
||||
|
||||
with patch.object(proxy_module, "llm_router", router):
|
||||
with patch(
|
||||
"litellm.router_utils.cooldown_handlers._set_cooldown_deployments"
|
||||
) as mock_cooldown:
|
||||
_write_health_state_to_router_cache(
|
||||
healthy_endpoints=[],
|
||||
unhealthy_endpoints=unhealthy_endpoints,
|
||||
)
|
||||
mock_cooldown.assert_not_called()
|
||||
|
||||
def test_401_still_triggers_cooldown_when_flag_enabled(self):
|
||||
"""Auth errors (401) still trigger cooldown even when flag is set."""
|
||||
import litellm.proxy.proxy_server as proxy_module
|
||||
from litellm.proxy.proxy_server import _write_health_state_to_router_cache
|
||||
|
||||
router = Router(
|
||||
model_list=[_make_model("deploy-1"), _make_model("deploy-2", "gpt-5")],
|
||||
allowed_fails_policy=AllowedFailsPolicy(AuthenticationErrorAllowedFails=0),
|
||||
enable_health_check_routing=True,
|
||||
health_check_ignore_transient_errors=True,
|
||||
)
|
||||
|
||||
auth_exc = litellm.AuthenticationError(
|
||||
message="Invalid key", model="gpt-4", llm_provider="openai"
|
||||
)
|
||||
|
||||
unhealthy_endpoints = [
|
||||
{"model_id": "deploy-1", "error": "auth failed", "exception": auth_exc},
|
||||
]
|
||||
|
||||
with patch.object(proxy_module, "llm_router", router):
|
||||
with patch(
|
||||
"litellm.router_utils.cooldown_handlers._set_cooldown_deployments"
|
||||
) as mock_cooldown:
|
||||
_write_health_state_to_router_cache(
|
||||
healthy_endpoints=[],
|
||||
unhealthy_endpoints=unhealthy_endpoints,
|
||||
)
|
||||
mock_cooldown.assert_called_once()
|
||||
|
||||
def test_429_triggers_cooldown_when_flag_disabled(self):
|
||||
"""When flag is False (default), 429 still triggers cooldown."""
|
||||
import litellm.proxy.proxy_server as proxy_module
|
||||
from litellm.proxy.proxy_server import _write_health_state_to_router_cache
|
||||
|
||||
router = Router(
|
||||
model_list=[_make_model("deploy-1"), _make_model("deploy-2", "gpt-5")],
|
||||
allowed_fails_policy=AllowedFailsPolicy(RateLimitErrorAllowedFails=0),
|
||||
enable_health_check_routing=True,
|
||||
health_check_ignore_transient_errors=False,
|
||||
)
|
||||
|
||||
rate_exc = litellm.RateLimitError(
|
||||
message="Rate limited", model="gpt-4", llm_provider="openai"
|
||||
)
|
||||
|
||||
unhealthy_endpoints = [
|
||||
{"model_id": "deploy-1", "error": "rate limited", "exception": rate_exc},
|
||||
]
|
||||
|
||||
with patch.object(proxy_module, "llm_router", router):
|
||||
with patch(
|
||||
"litellm.router_utils.cooldown_handlers._set_cooldown_deployments"
|
||||
) as mock_cooldown:
|
||||
_write_health_state_to_router_cache(
|
||||
healthy_endpoints=[],
|
||||
unhealthy_endpoints=unhealthy_endpoints,
|
||||
)
|
||||
mock_cooldown.assert_called_once()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue