fix(health-check-routing): gate cooldown loop behind allowed_fails_policy

Without the policy, cooldown is not the routing exclusion mechanism.
Firing _set_cooldown_deployments for all enable_health_check_routing users
was a backwards-incompatible change — 401s would immediately cooldown
deployments that the binary filter would have recovered on the next cycle.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Sameer Kankute 2026-04-02 19:44:19 +05:30
parent 92df1e77fa
commit aa17af2a37
No known key found for this signature in database

View file

@ -2163,35 +2163,40 @@ def _write_health_state_to_router_cache(
sum(1 for s in states.values() if not s.get("is_healthy")),
)
for endpoint in unhealthy_endpoints:
model_id = endpoint.get("model_id")
if not model_id:
continue
# Only feed health check failures into the cooldown pipeline when
# allowed_fails_policy is configured. Without it, cooldown is not the
# routing exclusion mechanism and triggering it would be a
# backwards-incompatible behavioral change for existing users.
if llm_router.allowed_fails_policy is not None:
for endpoint in unhealthy_endpoints:
model_id = endpoint.get("model_id")
if not model_id:
continue
original_exception = _exceptions.get(model_id)
if original_exception is None:
continue
original_exception = _exceptions.get(model_id)
if original_exception is None:
continue
exception_status = getattr(original_exception, "status_code", 500)
exception_status = getattr(original_exception, "status_code", 500)
if llm_router.health_check_ignore_transient_errors and exception_status in (
429,
408,
):
continue
if llm_router.health_check_ignore_transient_errors and exception_status in (
429,
408,
):
continue
increment_deployment_failures_for_current_minute(
litellm_router_instance=llm_router,
deployment_id=model_id,
)
increment_deployment_failures_for_current_minute(
litellm_router_instance=llm_router,
deployment_id=model_id,
)
_set_cooldown_deployments(
litellm_router_instance=llm_router,
original_exception=original_exception,
exception_status=exception_status,
deployment=model_id,
time_to_cooldown=llm_router.cooldown_time,
)
_set_cooldown_deployments(
litellm_router_instance=llm_router,
original_exception=original_exception,
exception_status=exception_status,
deployment=model_id,
time_to_cooldown=llm_router.cooldown_time,
)
except Exception as e:
verbose_proxy_logger.warning(