From 3fad8b2f14875c854e1c4a7b334e4a97fa753e5f Mon Sep 17 00:00:00 2001 From: Rene Zander Date: Fri, 24 Jul 2026 11:52:03 +0000 Subject: [PATCH 1/3] fix(proxy): stop health-check log storm for offline deployments Background health checks emit a full connection-error traceback on every poll cycle when a deployment's host is offline (#34281). Two changes in the health-check path: - Set no-log on internal health-check probes so their failures skip user logging callbacks (proxy cost/DB callbacks still run), removing the per-cycle traceback storm at its source. - Track reachability per deployment and log on transition only: one WARNING on healthy to unhealthy, one INFO on unhealthy to healthy, with a cooldown so a host that is offline by design does not re-emit every interval. Transport errors (connection refused, DNS, TLS, timeout) are classified as a degraded state and logged concisely; real provider errors keep detail. Fixes #34281 --- .../health_check_helpers.py | 9 + litellm/proxy/health_check.py | 157 ++++++++++++++++ .../test_health_check_transition_logging.py | 173 ++++++++++++++++++ 3 files changed, 339 insertions(+) create mode 100644 tests/test_litellm/proxy/test_health_check_transition_logging.py diff --git a/litellm/litellm_core_utils/health_check_helpers.py b/litellm/litellm_core_utils/health_check_helpers.py index 9fc036e2a99..3a4035e973a 100644 --- a/litellm/litellm_core_utils/health_check_helpers.py +++ b/litellm/litellm_core_utils/health_check_helpers.py @@ -54,6 +54,14 @@ class HealthCheckHelpers: 1. `tags`: This helps identify health check calls in the DB. 2. `user_api_key_auth`: This helps identify health check calls in the DB. We need this since the DB requires an API Key to track a log in the SpendLogs Table + 3. `no-log`: Health checks are infrastructure probes, not user + traffic, so their requests must not be routed through user + logging integrations. Without this, a deployment whose host is + offline emits a full connection-error traceback through the + logging callbacks on every poll cycle (see issue #34281). + `no-log` skips those user callbacks while proxy cost/DB + callbacks still run (see Logging.should_run_callback), so + health state is still recorded. """ from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup @@ -61,6 +69,7 @@ class HealthCheckHelpers: _metadata_variable_name = "litellm_metadata" litellm_metadata = HealthCheckHelpers._get_metadata_for_health_check_call() model_params[_metadata_variable_name] = litellm_metadata + model_params["no-log"] = True model_params = LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata( data=model_params, user_api_key_dict=UserAPIKeyAuth.get_litellm_internal_health_check_user_api_key_auth(), diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index 29c023df1a6..f278dfdb59e 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -9,6 +9,7 @@ import time from collections.abc import Mapping from typing import List, Optional +import httpx import litellm logger = logging.getLogger(__name__) @@ -131,6 +132,152 @@ def _clean_endpoint_data(endpoint_data: dict, details: Optional[bool] = True): ) +# --------------------------------------------------------------------------- +# Deployment reachability state tracking (issue #34281) +# +# A deployment whose host is offline used to emit a full stack trace on every +# background health-check poll. We track reachability per deployment id and log +# a single line when the state changes: one WARNING on healthy -> unhealthy and +# one INFO on unhealthy -> healthy. A still-unhealthy deployment is not re-logged +# every cycle, so an ad-hoc host that is offline by design stays quiet until it +# recovers. +# --------------------------------------------------------------------------- + +# deployment_id -> {"reachable": bool, "since": float, "last_logged": float} +_deployment_reachability_state: dict = {} + +# Re-emit a one-line "still unreachable" WARNING at most once per this many +# seconds while a deployment stays down. 0 (default) means log on transition +# only. Override via ``litellm.health_check_unreachable_relog_seconds``. +_DEFAULT_UNREACHABLE_RELOG_SECONDS = 0.0 + + +def _health_unreachable_relog_seconds() -> float: + value = getattr(litellm, "health_check_unreachable_relog_seconds", None) + if value is None: + return _DEFAULT_UNREACHABLE_RELOG_SECONDS + try: + return max(0.0, float(value)) + except (TypeError, ValueError): + return _DEFAULT_UNREACHABLE_RELOG_SECONDS + + +def _deployment_label(endpoint: dict) -> str: + """Human-readable identifier for a deployment in health-check logs.""" + model = endpoint.get("model") or endpoint.get("model_name") or endpoint.get("model_id") or "unknown" + api_base = endpoint.get("api_base") + return f"{model} ({api_base})" if api_base else str(model) + + +def _short_error(exc: BaseException | None) -> str: + """First line of an exception, bounded, for a single-line health log.""" + if exc is None: + return "unknown error" + text = str(exc).strip() + if not text: + return exc.__class__.__name__ + return text.splitlines()[0][:200] + + +def _is_transport_error(exc: BaseException | None) -> bool: + """ + True when the failure means the provider is unreachable (connection refused, + DNS failure, TLS error, timeout) rather than a real error returned by the + provider. An unreachable host is a degraded state, so we log it as a concise + WARNING; a real provider error keeps its full detail. + """ + if exc is None: + return False + if isinstance(exc, (ConnectionError, TimeoutError, OSError, httpx.TransportError)): + return True + for attr in ("APIConnectionError", "Timeout", "ServiceUnavailableError"): + litellm_exc = getattr(litellm, attr, None) + if litellm_exc is not None and isinstance(exc, litellm_exc): + return True + text = f"{exc.__class__.__name__} {exc}".lower() + return any( + signature in text + for signature in ( + "connection refused", + "connect call failed", + "cannot connect", + "name or service not known", + "temporary failure in name resolution", + "network is unreachable", + "no route to host", + "timed out", + "timeout", + ) + ) + + +def _log_deployment_health_transitions( + healthy_endpoints: list, + unhealthy_endpoints: list, + exceptions_by_model_id: dict, +) -> None: + """ + Log one line per deployment reachability transition instead of a stack trace + per poll cycle. Intended for the background health loop. Never raises. + """ + now = time.monotonic() + relog_seconds = _health_unreachable_relog_seconds() + + for endpoint in healthy_endpoints: + model_id = endpoint.get("model_id") + if not model_id: + continue + previous = _deployment_reachability_state.get(model_id) + if previous is not None and not previous.get("reachable", True): + logger.info( + "health_check: deployment %s is reachable again", + _deployment_label(endpoint), + ) + _deployment_reachability_state[model_id] = { + "reachable": True, + "since": now, + "last_logged": now, + } + + for endpoint in unhealthy_endpoints: + model_id = endpoint.get("model_id") + if not model_id: + continue + exc = exceptions_by_model_id.get(model_id) + previous = _deployment_reachability_state.get(model_id) + is_transition = previous is None or previous.get("reachable", True) + + if is_transition: + if _is_transport_error(exc): + logger.warning( + "health_check: deployment %s is unreachable (%s); suppressing per-cycle logs until it recovers", + _deployment_label(endpoint), + _short_error(exc), + ) + else: + logger.warning( + "health_check: deployment %s failed its health check: %s", + _deployment_label(endpoint), + _short_error(exc), + ) + _deployment_reachability_state[model_id] = { + "reachable": False, + "since": now, + "last_logged": now, + } + else: + last_logged = previous.get("last_logged", previous.get("since", now)) + if relog_seconds and (now - last_logged) >= relog_seconds: + logger.warning( + "health_check: deployment %s still unreachable after %.0fs", + _deployment_label(endpoint), + now - previous.get("since", now), + ) + previous["last_logged"] = now + previous["reachable"] = False + _deployment_reachability_state[model_id] = previous + + def health_check_filter_kwargs_from_general_settings( general_settings: Optional[dict], ) -> dict: @@ -648,4 +795,14 @@ async def perform_health_check( _rss_mb_for_log(), ) + # Emit one log line per reachability transition (down/up) for the recurring + # background poll, instead of a stack trace per cycle (issue #34281). Gated + # to the background loop so an on-demand /health call does not mutate the + # shared reachability state. Never allowed to break the health cycle. + if source == "proxy_background_loop": + try: + _log_deployment_health_transitions(healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id) + except Exception: # noqa: BLE001 + logger.debug("health_check: transition logging failed", exc_info=True) + return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id diff --git a/tests/test_litellm/proxy/test_health_check_transition_logging.py b/tests/test_litellm/proxy/test_health_check_transition_logging.py new file mode 100644 index 00000000000..6579f201fc4 --- /dev/null +++ b/tests/test_litellm/proxy/test_health_check_transition_logging.py @@ -0,0 +1,173 @@ +""" +Tests for health-check reachability transition logging and the no-log flag on +internal health probes (issue #34281): an offline deployment should produce a +single log line per state change instead of a full stack trace per poll cycle, +and health probes should not be routed through user logging callbacks. +""" + +import logging +import os +import sys + +import httpx +import pytest + +sys.path.insert(0, os.path.abspath("../../..")) + +import litellm +from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers +from litellm.proxy import health_check as hc + + +@pytest.fixture(autouse=True) +def _reset_reachability_state(): + """Isolate the module-level reachability state between tests.""" + hc._deployment_reachability_state.clear() + if hasattr(litellm, "health_check_unreachable_relog_seconds"): + delattr(litellm, "health_check_unreachable_relog_seconds") + yield + hc._deployment_reachability_state.clear() + + +# --------------------------------------------------------------------------- +# no-log flag on internal health probes +# --------------------------------------------------------------------------- + + +def test_health_check_tracking_sets_no_log(): + """Health probes must be marked no-log so failures skip user logging + callbacks (proxy cost/DB callbacks still run).""" + updated = HealthCheckHelpers._update_model_params_with_health_check_tracking_information( + model_params={"model": "gpt-4", "api_base": "http://localhost:1234"} + ) + assert updated["no-log"] is True + + +# --------------------------------------------------------------------------- +# transport-error classification +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "exc,expected", + [ + (ConnectionRefusedError("[Errno 111] Connection refused"), True), + (httpx.ConnectError("Cannot connect to host"), True), + (httpx.ConnectTimeout("timed out"), True), + (TimeoutError(), True), + (OSError("Network is unreachable"), True), + (Exception("AuthenticationError: invalid api key - status 401"), False), + (Exception("BadRequestError: unsupported parameter"), False), + (None, False), + ], +) +def test_is_transport_error_classification(exc, expected): + assert hc._is_transport_error(exc) is expected + + +# --------------------------------------------------------------------------- +# transition logging: down -> quiet -> up +# --------------------------------------------------------------------------- + + +def _unhealthy(model_id="d1"): + return [{"model_id": model_id, "model": "qwen3", "api_base": "http://ollama:8443"}] + + +def _healthy(model_id="d1"): + return [{"model_id": model_id, "model": "qwen3", "api_base": "http://ollama:8443"}] + + +def _warnings(caplog): + return [r for r in caplog.records if r.levelno == logging.WARNING] + + +def _infos(caplog): + return [r for r in caplog.records if r.levelno == logging.INFO] + + +def test_first_failure_logs_single_warning(caplog): + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + exc = ConnectionRefusedError("[Errno 111] Connection refused") + + hc._log_deployment_health_transitions( + healthy_endpoints=[], + unhealthy_endpoints=_unhealthy(), + exceptions_by_model_id={"d1": exc}, + ) + + warnings = _warnings(caplog) + assert len(warnings) == 1 + msg = warnings[0].getMessage() + assert "unreachable" in msg + assert "qwen3" in msg and "http://ollama:8443" in msg + assert hc._deployment_reachability_state["d1"]["reachable"] is False + + +def test_still_unhealthy_does_not_relog(caplog): + exc = ConnectionRefusedError("[Errno 111] Connection refused") + # cycle 1: transition down + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + # cycle 2: still down -> must be silent with default relog (0) + caplog.clear() + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + assert _warnings(caplog) == [] + + +def test_recovery_logs_single_info(caplog): + exc = ConnectionRefusedError("[Errno 111] Connection refused") + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + caplog.clear() + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + hc._log_deployment_health_transitions(_healthy(), [], {}) + + infos = _infos(caplog) + assert len(infos) == 1 + assert "reachable again" in infos[0].getMessage() + assert hc._deployment_reachability_state["d1"]["reachable"] is True + + +def test_real_error_keeps_detail(caplog): + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + exc = Exception("AuthenticationError: invalid api key - status 401") + + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + warnings = _warnings(caplog) + assert len(warnings) == 1 + msg = warnings[0].getMessage() + assert "failed its health check" in msg + assert "invalid api key" in msg + + +def test_relog_cooldown_reemits_after_interval(caplog): + litellm.health_check_unreachable_relog_seconds = 1 + exc = ConnectionRefusedError("[Errno 111] Connection refused") + # transition down + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + # simulate the cooldown having elapsed + hc._deployment_reachability_state["d1"]["last_logged"] -= 100 + + caplog.clear() + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + hc._log_deployment_health_transitions([], _unhealthy(), {"d1": exc}) + + warnings = _warnings(caplog) + assert len(warnings) == 1 + assert "still unreachable" in warnings[0].getMessage() + + +def test_endpoint_without_model_id_is_ignored(caplog): + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + hc._log_deployment_health_transitions( + healthy_endpoints=[], + unhealthy_endpoints=[{"model": "no-id"}], + exceptions_by_model_id={}, + ) + assert _warnings(caplog) == [] + assert hc._deployment_reachability_state == {} From bd52f1ded8fa67412553061fa90eba924ad89e93 Mon Sep 17 00:00:00 2001 From: Rene Zander Date: Fri, 24 Jul 2026 12:53:25 +0000 Subject: [PATCH 2/3] fix(proxy): keep perform_health_check under complexity limit Move the background-loop gate and error guard for transition logging into a small _maybe_log_health_transitions helper so perform_health_check gains only a single call, keeping it under the strict C901 ceiling. No behavior change. --- litellm/proxy/health_check.py | 29 ++++++++++++++----- .../test_health_check_transition_logging.py | 15 ++++++++++ 2 files changed, 36 insertions(+), 8 deletions(-) diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index f278dfdb59e..0c22d3ae35a 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -278,6 +278,25 @@ def _log_deployment_health_transitions( _deployment_reachability_state[model_id] = previous +def _maybe_log_health_transitions( + source: str, + healthy_endpoints: list, + unhealthy_endpoints: list, + exceptions_by_model_id: dict, +) -> None: + """ + Gate transition logging to the recurring background poll (so an on-demand + /health call does not mutate the shared reachability state) and never let a + logging error break the health cycle. + """ + if source != "proxy_background_loop": + return + try: + _log_deployment_health_transitions(healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id) + except Exception: # noqa: BLE001 + logger.debug("health_check: transition logging failed", exc_info=True) + + def health_check_filter_kwargs_from_general_settings( general_settings: Optional[dict], ) -> dict: @@ -796,13 +815,7 @@ async def perform_health_check( ) # Emit one log line per reachability transition (down/up) for the recurring - # background poll, instead of a stack trace per cycle (issue #34281). Gated - # to the background loop so an on-demand /health call does not mutate the - # shared reachability state. Never allowed to break the health cycle. - if source == "proxy_background_loop": - try: - _log_deployment_health_transitions(healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id) - except Exception: # noqa: BLE001 - logger.debug("health_check: transition logging failed", exc_info=True) + # background poll, instead of a stack trace per cycle (issue #34281). + _maybe_log_health_transitions(source, healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id) return healthy_endpoints, unhealthy_endpoints, exceptions_by_model_id diff --git a/tests/test_litellm/proxy/test_health_check_transition_logging.py b/tests/test_litellm/proxy/test_health_check_transition_logging.py index 6579f201fc4..034c7db6787 100644 --- a/tests/test_litellm/proxy/test_health_check_transition_logging.py +++ b/tests/test_litellm/proxy/test_health_check_transition_logging.py @@ -162,6 +162,21 @@ def test_relog_cooldown_reemits_after_interval(caplog): assert "still unreachable" in warnings[0].getMessage() +def test_maybe_log_gated_to_background_loop(caplog): + caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") + exc = ConnectionRefusedError("[Errno 111] Connection refused") + + # on-demand /health source: must not log or touch shared state + hc._maybe_log_health_transitions("endpoint", [], _unhealthy(), {"d1": exc}) + assert _warnings(caplog) == [] + assert hc._deployment_reachability_state == {} + + # background loop source: logs and records state + hc._maybe_log_health_transitions("proxy_background_loop", [], _unhealthy(), {"d1": exc}) + assert len(_warnings(caplog)) == 1 + assert hc._deployment_reachability_state["d1"]["reachable"] is False + + def test_endpoint_without_model_id_is_ignored(caplog): caplog.set_level(logging.INFO, logger="litellm.proxy.health_check") hc._log_deployment_health_transitions( From 45590af88b1159d9527dcbe56bdd851a314a093b Mon Sep 17 00:00:00 2001 From: Rene Zander Date: Fri, 24 Jul 2026 13:21:05 +0000 Subject: [PATCH 3/3] fix(proxy): narrow Optional state dict in transition logging Inline the None-check into the branch condition so basedpyright narrows the reachability state dict in the else branch, avoiding reportOptionalSubscript on the state updates. No behavior change. --- litellm/proxy/health_check.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index 0c22d3ae35a..e01d55e28e2 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -245,9 +245,10 @@ def _log_deployment_health_transitions( continue exc = exceptions_by_model_id.get(model_id) previous = _deployment_reachability_state.get(model_id) - is_transition = previous is None or previous.get("reachable", True) - if is_transition: + # `previous is None` in the condition narrows `previous` to a dict in the + # else branch below, so the state updates there are not Optional access. + if previous is None or previous.get("reachable", True): if _is_transport_error(exc): logger.warning( "health_check: deployment %s is unreachable (%s); suppressing per-cycle logs until it recovers",