fix(router): also exclude 429/408 from health state cache when ignore_transient_errors set

The previous fix only skipped cooldown counter increments. The health state
cache was still marking 429/408 endpoints as is_healthy=False, causing the
binary health check filter to exclude them from routing.

Now, when health_check_ignore_transient_errors=True, 429/408 endpoints are
also excluded from the unhealthy list passed to build_deployment_health_states(),
so the binary filter treats them as unaffected (not unhealthy).

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Sameer Kankute 2026-04-02 17:02:48 +05:30 • committed by Yuneng Jiang
parent 4fb8932e36
commit 9a605e2bc5
No known key found for this signature in database
2 changed files with 42 additions and 1 deletions

View file

@ -2264,9 +2264,19 @@ def _write_health_state_to_router_cache(
if llm_router is None or not llm_router.enable_health_check_routing:
return
# When health_check_ignore_transient_errors is set, treat 429/408
# endpoints as healthy so they are not filtered from routing.
_effective_unhealthy = unhealthy_endpoints
if llm_router.health_check_ignore_transient_errors:
_effective_unhealthy = [
ep
for ep in unhealthy_endpoints
if getattr(ep.get("exception"), "status_code", 500) not in (429, 408)
]
states = build_deployment_health_states(
healthy_endpoints=healthy_endpoints,
unhealthy_endpoints=unhealthy_endpoints,
unhealthy_endpoints=_effective_unhealthy,
)
if states:
llm_router.health_state_cache.set_deployment_health_states(states)

View file

@ -634,6 +634,37 @@ class TestHealthCheckIgnoreTransientErrors:
)
mock_cooldown.assert_called_once()
def test_429_not_written_to_health_state_cache_when_flag_enabled(self):
"""429 endpoint is excluded from health state cache when flag is set,
so the binary health check filter does not mark it as unhealthy."""
import litellm.proxy.proxy_server as proxy_module
from litellm.proxy.proxy_server import _write_health_state_to_router_cache
router = Router(
model_list=[_make_model("deploy-1")],
enable_health_check_routing=True,
health_check_ignore_transient_errors=True,
)
rate_exc = litellm.RateLimitError(
message="Rate limited", model="gpt-4", llm_provider="openai"
)
unhealthy_endpoints = [
{"model_id": "deploy-1", "error": "rate limited", "exception": rate_exc},
]
with patch.object(proxy_module, "llm_router", router):
_write_health_state_to_router_cache(
healthy_endpoints=[],
unhealthy_endpoints=unhealthy_endpoints,
)
# Health state cache should have NO entry for deploy-1
# (429 was ignored, not written as unhealthy)
unhealthy_ids = router.health_state_cache.get_unhealthy_deployment_ids()
assert "deploy-1" not in unhealthy_ids
def test_429_triggers_cooldown_when_flag_disabled(self):
"""When flag is False (default), 429 still triggers cooldown."""
import litellm.proxy.proxy_server as proxy_module