From 79be436c2bbc6c259de6094aaff9eb7699495882 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 31 Jul 2025 13:48:35 -0700 Subject: [PATCH] [Feat] Background Health Checks - Allow disabling background health checks for a specific (#13186) * disable background health checks for specific models * test_background_health_check_skip_disabled_models * Disable Background Health Checks For Specific Models --- docs/my-website/docs/proxy/health.md | 21 +++++++++++-- litellm/proxy/proxy_server.py | 7 +++++ tests/proxy_unit_tests/test_proxy_server.py | 35 +++++++++++++++++++++ 3 files changed, 61 insertions(+), 2 deletions(-) diff --git a/docs/my-website/docs/proxy/health.md b/docs/my-website/docs/proxy/health.md index 52321a38457..6788045a0a2 100644 --- a/docs/my-website/docs/proxy/health.md +++ b/docs/my-website/docs/proxy/health.md @@ -219,7 +219,7 @@ Here's how to use it: ``` general_settings: background_health_checks: True # enable background health checks - health_check_interval: 300 # frequency of background health checks + health_check_interval: 300 # frequency of background health checks ``` 2. Start server @@ -229,7 +229,24 @@ $ litellm /path/to/config.yaml 3. Query health endpoint: ``` -curl --location 'http://0.0.0.0:4000/health' + curl --location 'http://0.0.0.0:4000/health' +``` + +### Disable Background Health Checks For Specific Models + +Use this if you want to disable background health checks for specific models. + +If `background_health_checks` is enabled you can skip individual models by +setting `disable_background_health_check: true` in the model's `model_info`. + +```yaml +model_list: + - model_name: openai/gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + model_info: + disable_background_health_check: true ``` ### Hide details diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index a95a8a8835f..6b4b4c25dd0 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -1353,6 +1353,13 @@ async def _run_background_health_check(): # make 1 deep copy of llm_model_list on every health check iteration _llm_model_list = copy.deepcopy(llm_model_list) or [] + # filter out models that have disabled background health checks + _llm_model_list = [ + m + for m in _llm_model_list + if not m.get("model_info", {}).get("disable_background_health_check", False) + ] + healthy_endpoints, unhealthy_endpoints = await perform_health_check( model_list=_llm_model_list, details=health_check_details ) diff --git a/tests/proxy_unit_tests/test_proxy_server.py b/tests/proxy_unit_tests/test_proxy_server.py index 3438cf6fb21..139c90fd3cb 100644 --- a/tests/proxy_unit_tests/test_proxy_server.py +++ b/tests/proxy_unit_tests/test_proxy_server.py @@ -2250,6 +2250,41 @@ async def test_run_background_health_check_reflects_llm_model_list(monkeypatch): assert called_model_lists[1] == test_model_list_2 +@pytest.mark.asyncio +async def test_background_health_check_skip_disabled_models(monkeypatch): + """Ensure models with disable_background_health_check are skipped.""" + import litellm.proxy.proxy_server as proxy_server + import copy + + test_model_list = [ + {"model_name": "model-a"}, + {"model_name": "model-b", "model_info": {"disable_background_health_check": True}}, + ] + called_model_lists = [] + + async def fake_perform_health_check(model_list, details): + called_model_lists.append(copy.deepcopy(model_list)) + return (["healthy"], []) + + monkeypatch.setattr(proxy_server, "health_check_interval", 1) + monkeypatch.setattr(proxy_server, "health_check_details", None) + monkeypatch.setattr(proxy_server, "llm_model_list", copy.deepcopy(test_model_list)) + monkeypatch.setattr(proxy_server, "perform_health_check", fake_perform_health_check) + monkeypatch.setattr(proxy_server, "health_check_results", {}) + + async def fake_sleep(interval): + raise asyncio.CancelledError() + + monkeypatch.setattr(asyncio, "sleep", fake_sleep) + + try: + await proxy_server._run_background_health_check() + except asyncio.CancelledError: + pass + + assert called_model_lists == [[{"model_name": "model-a"}]] + + def test_get_timeout_from_request(): from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup