add health backlog endpoint wiring and coverage

Wire BacklogHealth into health endpoints by adding a dedicated /health/backlog route and validate parsing/caching behavior with focused proxy health endpoint tests.

Made-with: Cursor
This commit is contained in:
Ishaan Jaffer 2026-02-26 14:08:05 -08:00
parent 98df8a6bb7
commit 3b04839c9e
2 changed files with 99 additions and 1 deletions

View file

@ -33,6 +33,7 @@ from litellm.proxy.health_check import (
perform_health_check,
run_with_timeout,
)
from litellm.proxy.health_endpoints.backlog_health import BacklogHealth
from litellm.secret_managers.main import get_secret
#### Health ENDPOINTS ####
@ -1130,7 +1131,6 @@ async def _db_health_readiness_check():
PrismaDBExceptionHandler.handle_db_exception(e)
return db_health_cache
@router.get(
"/settings",
tags=["health"],
@ -1297,6 +1297,20 @@ async def health_readiness():
raise HTTPException(status_code=503, detail=f"Service Unhealthy ({str(e)})")
@router.get(
"/health/backlog",
tags=["health"],
dependencies=[Depends(user_api_key_auth)],
)
async def health_backlog(
port: Optional[int] = fastapi.Query(
None, description="Port to inspect. Defaults to PORT env var or 4000."
),
):
resolved_port = port or int(os.getenv("PORT", "4000"))
return BacklogHealth.get_cached_listen_queue_stats(port=resolved_port)
@router.get(
"/health/liveliness", # Historical LiteLLM name; doesn't match k8s terminology but kept for backwards compatibility
tags=["health"],
@ -1329,6 +1343,23 @@ async def health_readiness_options():
return Response(headers=response_headers, status_code=200)
@router.options(
"/health/backlog",
tags=["health"],
dependencies=[Depends(user_api_key_auth)],
)
async def health_backlog_options():
"""
Options endpoint for health/backlog check.
"""
response_headers = {
"Allow": "GET, OPTIONS",
"Access-Control-Allow-Methods": "GET, OPTIONS",
"Access-Control-Allow-Headers": "*",
}
return Response(headers=response_headers, status_code=200)
@router.options(
"/health/liveliness",
tags=["health"],

View file

@ -23,6 +23,7 @@ from litellm.proxy.health_endpoints._health_endpoints import (
from litellm.proxy.health_endpoints._health_endpoints import (
test_model_connection as health_test_model_connection,
)
from litellm.proxy.health_endpoints.backlog_health import BacklogHealth
# Import shared proxy test helpers from conftest
from tests.test_litellm.proxy.conftest import create_proxy_test_client
@ -481,6 +482,72 @@ def test_health_readiness(proxy_client):
print("="*60 + "\n")
def test_parse_linux_ss_listen_queue():
sample_output = """State Recv-Q Send-Q Local Address:Port Peer Address:Port
LISTEN 3 128 0.0.0.0:4000 0.0.0.0:*
"""
result = BacklogHealth.parse_linux_ss_listen_queue(output=sample_output, port=4000)
assert result["listen_queue_current"] == 3
assert result["listen_queue_max"] == 128
def test_parse_macos_netstat_listen_queue():
sample_output = """Current listen queue sizes (qlen/incqlen/maxqlen)
tcp4 2/0/128 *.4000 *.* LISTEN
"""
result = BacklogHealth.parse_macos_netstat_listen_queue(
output=sample_output, port=4000
)
assert result["listen_queue_current"] == 2
assert result["listen_queue_max"] == 128
def test_get_cached_listen_queue_stats_uses_cache():
mock_stats = {
"status": "healthy",
"port": 4000,
"listen_queue_current": 1,
"listen_queue_max": 128,
"listen_queue_utilization": 0.0078,
"listen_queue_status": "ok",
"sampled_at": "2026-02-26T00:00:00Z",
}
BacklogHealth.cache = {"last_updated": datetime.min, "port": None, "data": None}
with patch(
"litellm.proxy.health_endpoints.backlog_health.BacklogHealth.read_listen_queue_stats_for_port",
return_value=mock_stats,
) as mock_reader:
first = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30)
second = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30)
assert first == second
assert mock_reader.call_count == 1
def test_health_backlog_endpoint(proxy_client):
mock_stats = {
"status": "healthy",
"port": 4000,
"source": "ss",
"listen_queue_current": 7,
"listen_queue_max": 128,
"listen_queue_utilization": 0.0547,
"listen_queue_status": "ok",
"sampled_at": "2026-02-26T00:00:00Z",
}
with patch(
"litellm.proxy.health_endpoints._health_endpoints.BacklogHealth.get_cached_listen_queue_stats",
return_value=mock_stats,
):
response = proxy_client.get("/health/backlog")
assert response.status_code == 200
payload = response.json()
assert payload["status"] == "healthy"
assert payload["listen_queue_current"] == 7
assert payload["listen_queue_max"] == 128
def test_get_callback_identifier_string_and_object_with_callback_name():
"""
Test get_callback_identifier with string callbacks and objects with callback_name attribute.