diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index 4496ad92631..0d06eae3ee7 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -33,6 +33,7 @@ from litellm.proxy.health_check import ( perform_health_check, run_with_timeout, ) +from litellm.proxy.health_endpoints.backlog_health import BacklogHealth from litellm.secret_managers.main import get_secret #### Health ENDPOINTS #### @@ -1130,7 +1131,6 @@ async def _db_health_readiness_check(): PrismaDBExceptionHandler.handle_db_exception(e) return db_health_cache - @router.get( "/settings", tags=["health"], @@ -1297,6 +1297,20 @@ async def health_readiness(): raise HTTPException(status_code=503, detail=f"Service Unhealthy ({str(e)})") +@router.get( + "/health/backlog", + tags=["health"], + dependencies=[Depends(user_api_key_auth)], +) +async def health_backlog( + port: Optional[int] = fastapi.Query( + None, description="Port to inspect. Defaults to PORT env var or 4000." + ), +): + resolved_port = port or int(os.getenv("PORT", "4000")) + return BacklogHealth.get_cached_listen_queue_stats(port=resolved_port) + + @router.get( "/health/liveliness", # Historical LiteLLM name; doesn't match k8s terminology but kept for backwards compatibility tags=["health"], @@ -1329,6 +1343,23 @@ async def health_readiness_options(): return Response(headers=response_headers, status_code=200) +@router.options( + "/health/backlog", + tags=["health"], + dependencies=[Depends(user_api_key_auth)], +) +async def health_backlog_options(): + """ + Options endpoint for health/backlog check. + """ + response_headers = { + "Allow": "GET, OPTIONS", + "Access-Control-Allow-Methods": "GET, OPTIONS", + "Access-Control-Allow-Headers": "*", + } + return Response(headers=response_headers, status_code=200) + + @router.options( "/health/liveliness", tags=["health"], diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index cc6302644a7..81ea6c12ebf 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -23,6 +23,7 @@ from litellm.proxy.health_endpoints._health_endpoints import ( from litellm.proxy.health_endpoints._health_endpoints import ( test_model_connection as health_test_model_connection, ) +from litellm.proxy.health_endpoints.backlog_health import BacklogHealth # Import shared proxy test helpers from conftest from tests.test_litellm.proxy.conftest import create_proxy_test_client @@ -481,6 +482,72 @@ def test_health_readiness(proxy_client): print("="*60 + "\n") +def test_parse_linux_ss_listen_queue(): + sample_output = """State Recv-Q Send-Q Local Address:Port Peer Address:Port +LISTEN 3 128 0.0.0.0:4000 0.0.0.0:* +""" + result = BacklogHealth.parse_linux_ss_listen_queue(output=sample_output, port=4000) + assert result["listen_queue_current"] == 3 + assert result["listen_queue_max"] == 128 + + +def test_parse_macos_netstat_listen_queue(): + sample_output = """Current listen queue sizes (qlen/incqlen/maxqlen) +tcp4 2/0/128 *.4000 *.* LISTEN +""" + result = BacklogHealth.parse_macos_netstat_listen_queue( + output=sample_output, port=4000 + ) + assert result["listen_queue_current"] == 2 + assert result["listen_queue_max"] == 128 + + +def test_get_cached_listen_queue_stats_uses_cache(): + mock_stats = { + "status": "healthy", + "port": 4000, + "listen_queue_current": 1, + "listen_queue_max": 128, + "listen_queue_utilization": 0.0078, + "listen_queue_status": "ok", + "sampled_at": "2026-02-26T00:00:00Z", + } + BacklogHealth.cache = {"last_updated": datetime.min, "port": None, "data": None} + with patch( + "litellm.proxy.health_endpoints.backlog_health.BacklogHealth.read_listen_queue_stats_for_port", + return_value=mock_stats, + ) as mock_reader: + first = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30) + second = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30) + + assert first == second + assert mock_reader.call_count == 1 + + +def test_health_backlog_endpoint(proxy_client): + mock_stats = { + "status": "healthy", + "port": 4000, + "source": "ss", + "listen_queue_current": 7, + "listen_queue_max": 128, + "listen_queue_utilization": 0.0547, + "listen_queue_status": "ok", + "sampled_at": "2026-02-26T00:00:00Z", + } + with patch( + "litellm.proxy.health_endpoints._health_endpoints.BacklogHealth.get_cached_listen_queue_stats", + return_value=mock_stats, + ): + response = proxy_client.get("/health/backlog") + + assert response.status_code == 200 + payload = response.json() + assert payload["status"] == "healthy" + assert payload["listen_queue_current"] == 7 + assert payload["listen_queue_max"] == 128 + + def test_get_callback_identifier_string_and_object_with_callback_name(): """ Test get_callback_identifier with string callbacks and objects with callback_name attribute.