mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
add health backlog endpoint wiring and coverage
Wire BacklogHealth into health endpoints by adding a dedicated /health/backlog route and validate parsing/caching behavior with focused proxy health endpoint tests. Made-with: Cursor
This commit is contained in:
parent
98df8a6bb7
commit
3b04839c9e
2 changed files with 99 additions and 1 deletions
|
|
@ -33,6 +33,7 @@ from litellm.proxy.health_check import (
|
|||
perform_health_check,
|
||||
run_with_timeout,
|
||||
)
|
||||
from litellm.proxy.health_endpoints.backlog_health import BacklogHealth
|
||||
from litellm.secret_managers.main import get_secret
|
||||
|
||||
#### Health ENDPOINTS ####
|
||||
|
|
@ -1130,7 +1131,6 @@ async def _db_health_readiness_check():
|
|||
PrismaDBExceptionHandler.handle_db_exception(e)
|
||||
return db_health_cache
|
||||
|
||||
|
||||
@router.get(
|
||||
"/settings",
|
||||
tags=["health"],
|
||||
|
|
@ -1297,6 +1297,20 @@ async def health_readiness():
|
|||
raise HTTPException(status_code=503, detail=f"Service Unhealthy ({str(e)})")
|
||||
|
||||
|
||||
@router.get(
|
||||
"/health/backlog",
|
||||
tags=["health"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
)
|
||||
async def health_backlog(
|
||||
port: Optional[int] = fastapi.Query(
|
||||
None, description="Port to inspect. Defaults to PORT env var or 4000."
|
||||
),
|
||||
):
|
||||
resolved_port = port or int(os.getenv("PORT", "4000"))
|
||||
return BacklogHealth.get_cached_listen_queue_stats(port=resolved_port)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/health/liveliness", # Historical LiteLLM name; doesn't match k8s terminology but kept for backwards compatibility
|
||||
tags=["health"],
|
||||
|
|
@ -1329,6 +1343,23 @@ async def health_readiness_options():
|
|||
return Response(headers=response_headers, status_code=200)
|
||||
|
||||
|
||||
@router.options(
|
||||
"/health/backlog",
|
||||
tags=["health"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
)
|
||||
async def health_backlog_options():
|
||||
"""
|
||||
Options endpoint for health/backlog check.
|
||||
"""
|
||||
response_headers = {
|
||||
"Allow": "GET, OPTIONS",
|
||||
"Access-Control-Allow-Methods": "GET, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "*",
|
||||
}
|
||||
return Response(headers=response_headers, status_code=200)
|
||||
|
||||
|
||||
@router.options(
|
||||
"/health/liveliness",
|
||||
tags=["health"],
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ from litellm.proxy.health_endpoints._health_endpoints import (
|
|||
from litellm.proxy.health_endpoints._health_endpoints import (
|
||||
test_model_connection as health_test_model_connection,
|
||||
)
|
||||
from litellm.proxy.health_endpoints.backlog_health import BacklogHealth
|
||||
|
||||
# Import shared proxy test helpers from conftest
|
||||
from tests.test_litellm.proxy.conftest import create_proxy_test_client
|
||||
|
|
@ -481,6 +482,72 @@ def test_health_readiness(proxy_client):
|
|||
print("="*60 + "\n")
|
||||
|
||||
|
||||
def test_parse_linux_ss_listen_queue():
|
||||
sample_output = """State Recv-Q Send-Q Local Address:Port Peer Address:Port
|
||||
LISTEN 3 128 0.0.0.0:4000 0.0.0.0:*
|
||||
"""
|
||||
result = BacklogHealth.parse_linux_ss_listen_queue(output=sample_output, port=4000)
|
||||
assert result["listen_queue_current"] == 3
|
||||
assert result["listen_queue_max"] == 128
|
||||
|
||||
|
||||
def test_parse_macos_netstat_listen_queue():
|
||||
sample_output = """Current listen queue sizes (qlen/incqlen/maxqlen)
|
||||
tcp4 2/0/128 *.4000 *.* LISTEN
|
||||
"""
|
||||
result = BacklogHealth.parse_macos_netstat_listen_queue(
|
||||
output=sample_output, port=4000
|
||||
)
|
||||
assert result["listen_queue_current"] == 2
|
||||
assert result["listen_queue_max"] == 128
|
||||
|
||||
|
||||
def test_get_cached_listen_queue_stats_uses_cache():
|
||||
mock_stats = {
|
||||
"status": "healthy",
|
||||
"port": 4000,
|
||||
"listen_queue_current": 1,
|
||||
"listen_queue_max": 128,
|
||||
"listen_queue_utilization": 0.0078,
|
||||
"listen_queue_status": "ok",
|
||||
"sampled_at": "2026-02-26T00:00:00Z",
|
||||
}
|
||||
BacklogHealth.cache = {"last_updated": datetime.min, "port": None, "data": None}
|
||||
with patch(
|
||||
"litellm.proxy.health_endpoints.backlog_health.BacklogHealth.read_listen_queue_stats_for_port",
|
||||
return_value=mock_stats,
|
||||
) as mock_reader:
|
||||
first = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30)
|
||||
second = BacklogHealth.get_cached_listen_queue_stats(port=4000, ttl_seconds=30)
|
||||
|
||||
assert first == second
|
||||
assert mock_reader.call_count == 1
|
||||
|
||||
|
||||
def test_health_backlog_endpoint(proxy_client):
|
||||
mock_stats = {
|
||||
"status": "healthy",
|
||||
"port": 4000,
|
||||
"source": "ss",
|
||||
"listen_queue_current": 7,
|
||||
"listen_queue_max": 128,
|
||||
"listen_queue_utilization": 0.0547,
|
||||
"listen_queue_status": "ok",
|
||||
"sampled_at": "2026-02-26T00:00:00Z",
|
||||
}
|
||||
with patch(
|
||||
"litellm.proxy.health_endpoints._health_endpoints.BacklogHealth.get_cached_listen_queue_stats",
|
||||
return_value=mock_stats,
|
||||
):
|
||||
response = proxy_client.get("/health/backlog")
|
||||
|
||||
assert response.status_code == 200
|
||||
payload = response.json()
|
||||
assert payload["status"] == "healthy"
|
||||
assert payload["listen_queue_current"] == 7
|
||||
assert payload["listen_queue_max"] == 128
|
||||
|
||||
|
||||
def test_get_callback_identifier_string_and_object_with_callback_name():
|
||||
"""
|
||||
Test get_callback_identifier with string callbacks and objects with callback_name attribute.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue