feat(agents/): add health check for agent endpoints before returning

filter out bad endpoints
This commit is contained in:
Krrish Dholakia 2026-03-05 18:22:25 -08:00
parent 49ddcba44a
commit d7c5afa75f
6 changed files with 537 additions and 601 deletions

View file

@ -257,6 +257,21 @@ LiteLLM follows the [A2A JSON-RPC 2.0 specification](https://github.com/google/A
}
```
## Agent Health Checks
LiteLLM can automatically filter out agents whose backends are unreachable. The `GET /v1/agents` endpoint only returns agents that are currently healthy.
- **Without background checks (default):** Health checks run inline every time `GET /v1/agents` is called.
- **With background checks:** A background loop periodically pings agents so the endpoint responds instantly.
```yaml
general_settings:
background_agent_health_checks: true # run health checks in the background
health_check_interval: 300 # interval in seconds (default: 300)
```
For more details, see [Background Agent Health Checks](./proxy/health#background-agent-health-checks).
## Agent Registry
Want to create a central registry so your team can discover what agents are available within your company?

View file

@ -132,6 +132,7 @@ general_settings:
global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up
infer_model_from_keys: true
background_health_checks: true
background_agent_health_checks: true # enable background health checks for A2A agents
health_check_interval: 300
alerting: ["slack", "email"]
alerting_threshold: 0
@ -228,6 +229,7 @@ router_settings:
| global_max_parallel_requests | integer | The max parallel requests allowed on the proxy overall |
| infer_model_from_keys | boolean | If true, infers the model from the provided keys |
| background_health_checks | boolean | If true, enables background health checks. [Doc on health checks](health) |
| background_agent_health_checks | boolean | If true, enables background health checks for A2A agents. When disabled, health checks run inline on `GET /v1/agents`. [Doc on health checks](health#background-agent-health-checks) |
| health_check_interval | integer | The interval for health checks in seconds [Doc on health checks](health) |
| alerting | array of strings | List of alerting methods [Doc on Slack Alerting](alerting) |
| alerting_threshold | integer | The threshold for triggering alerts [Doc on Slack Alerting](alerting) |

View file

@ -284,6 +284,30 @@ $ litellm /path/to/config.yaml
curl --location 'http://0.0.0.0:4000/health'
```
### Background Agent Health Checks
You can enable background health checks for A2A agents registered on the proxy. When enabled, a background loop periodically pings each agent's URL and tracks which agents are healthy. The `GET /v1/agents` endpoint then only returns agents that are currently reachable.
If `background_agent_health_checks` is **not** enabled, the health check runs inline every time `GET /v1/agents` is called.
**How it works:**
- Sends an HTTP `HEAD` request to each agent's `url` (from `agent_card_params`)
- Any response with status code < 500 (including `405 Method Not Allowed`) means the agent's server is live
- Connection errors or timeouts mark the agent as unhealthy
- Agents using the completion bridge (no URL, only `custom_llm_provider`) are always treated as healthy
```yaml
general_settings:
background_agent_health_checks: true # enable background agent health checks
health_check_interval: 300 # shared interval for all background health checks (seconds)
```
```bash
# Only healthy agents are returned
curl -X GET "http://0.0.0.0:4000/v1/agents" \
-H "Authorization: Bearer sk-1234"
```
### Disable Background Health Checks For Specific Models
Use this if you want to disable background health checks for specific models.

View file

@ -1,13 +1,12 @@
import hashlib
import json
from datetime import datetime, timezone
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Optional, Set
import litellm
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy.management_helpers.object_permission_utils import (
handle_update_object_permission_common,
)
from litellm.proxy.management_helpers.object_permission_utils import \
handle_update_object_permission_common
from litellm.proxy.utils import PrismaClient
from litellm.types.agents import AgentConfig, AgentResponse, PatchAgentRequest
@ -15,6 +14,7 @@ from litellm.types.agents import AgentConfig, AgentResponse, PatchAgentRequest
class AgentRegistry:
def __init__(self):
self.agent_list: List[AgentResponse] = []
self.healthy_agent_ids: Optional[Set[str]] = None
def reset_agent_list(self):
self.agent_list = []
@ -43,6 +43,43 @@ class AgentRegistry:
public_agent_list.append(agent)
return public_agent_list
async def run_health_check(self) -> None:
"""Ping each agent's URL with a HEAD request and update healthy_agent_ids."""
from litellm.llms.custom_httpx.http_handler import \
get_async_httpx_client
from litellm.types.llms.custom_http import httpxSpecialProvider
client = get_async_httpx_client(llm_provider=httpxSpecialProvider.A2A)
healthy: Set[str] = set()
for agent in self.agent_list:
url = agent.agent_card_params.get("url")
if not url:
# Completion-bridge agents (no URL) are always treated as healthy
if agent.litellm_params and agent.litellm_params.get(
"custom_llm_provider"
):
healthy.add(agent.agent_id)
continue
try:
resp = await client.client.head(url, timeout=10.0)
if resp.status_code < 500: # handle 405 Method Not Allowed
healthy.add(agent.agent_id)
except Exception:
pass # skip if unhealthy
self.healthy_agent_ids = healthy
def get_healthy_agent_list(
self, agent_names: Optional[List[str]] = None
) -> List[AgentResponse]:
"""Return only agents whose health check passed.
If no health data exists yet (healthy_agent_ids is None), returns all agents.
"""
agents = self.get_agent_list(agent_names=agent_names)
if self.healthy_agent_ids is None:
return agents
return [a for a in agents if a.agent_id in self.healthy_agent_ids]
def _create_agent_id(self, agent_config: AgentConfig) -> str:
return hashlib.sha256(
json.dumps(agent_config, sort_keys=True).encode()
@ -149,9 +186,13 @@ class AgentRegistry:
created_agent_dict = created_agent.model_dump()
if created_agent.object_permission is not None:
try:
created_agent_dict["object_permission"] = created_agent.object_permission.model_dump()
created_agent_dict["object_permission"] = (
created_agent.object_permission.model_dump()
)
except Exception:
created_agent_dict["object_permission"] = created_agent.object_permission.dict()
created_agent_dict["object_permission"] = (
created_agent.object_permission.dict()
)
return AgentResponse(**created_agent_dict) # type: ignore
except Exception as e:
raise Exception(f"Error adding agent to DB: {str(e)}")
@ -219,12 +260,10 @@ class AgentRegistry:
existing_object_permission_id = existing_agent.get(
"object_permission_id"
)
object_permission_id = (
await handle_update_object_permission_common(
agent_copy,
existing_object_permission_id,
prisma_client,
)
object_permission_id = await handle_update_object_permission_common(
agent_copy,
existing_object_permission_id,
prisma_client,
)
if object_permission_id is not None:
update_data["object_permission_id"] = object_permission_id
@ -241,9 +280,13 @@ class AgentRegistry:
patched_agent_dict = patched_agent.model_dump()
if patched_agent.object_permission is not None:
try:
patched_agent_dict["object_permission"] = patched_agent.object_permission.model_dump()
patched_agent_dict["object_permission"] = (
patched_agent.object_permission.model_dump()
)
except Exception:
patched_agent_dict["object_permission"] = patched_agent.object_permission.dict()
patched_agent_dict["object_permission"] = (
patched_agent.object_permission.dict()
)
return AgentResponse(**patched_agent_dict) # type: ignore
except Exception as e:
raise Exception(f"Error patching agent in DB: {str(e)}")
@ -298,12 +341,10 @@ class AgentRegistry:
else None
)
agent_copy = dict(agent)
object_permission_id = (
await handle_update_object_permission_common(
agent_copy,
existing_object_permission_id,
prisma_client,
)
object_permission_id = await handle_update_object_permission_common(
agent_copy,
existing_object_permission_id,
prisma_client,
)
if object_permission_id is not None:
update_data["object_permission_id"] = object_permission_id
@ -318,9 +359,13 @@ class AgentRegistry:
updated_agent_dict = updated_agent.model_dump()
if updated_agent.object_permission is not None:
try:
updated_agent_dict["object_permission"] = updated_agent.object_permission.model_dump()
updated_agent_dict["object_permission"] = (
updated_agent.object_permission.model_dump()
)
except Exception:
updated_agent_dict["object_permission"] = updated_agent.object_permission.dict()
updated_agent_dict["object_permission"] = (
updated_agent.object_permission.dict()
)
return AgentResponse(**updated_agent_dict) # type: ignore
except Exception as e:
raise Exception(f"Error updating agent in DB: {str(e)}")
@ -344,7 +389,9 @@ class AgentRegistry:
# object_permission is eagerly loaded via include above
if agent.object_permission is not None:
try:
agent_dict["object_permission"] = agent.object_permission.model_dump()
agent_dict["object_permission"] = (
agent.object_permission.model_dump()
)
except Exception:
agent_dict["object_permission"] = agent.object_permission.dict()
agents.append(agent_dict)

View file

@ -14,19 +14,16 @@ from fastapi import APIRouter, Depends, HTTPException, Request
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy._types import (CommonProxyErrors, LitellmUserRoles,
UserAPIKeyAuth)
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.management_endpoints.common_daily_activity import get_daily_activity
from litellm.types.agents import (
AgentConfig,
AgentMakePublicResponse,
AgentResponse,
MakeAgentsPublicRequest,
PatchAgentRequest,
)
from litellm.types.proxy.management_endpoints.common_daily_activity import (
SpendAnalyticsPaginatedResponse,
)
from litellm.proxy.management_endpoints.common_daily_activity import \
get_daily_activity
from litellm.types.agents import (AgentConfig, AgentMakePublicResponse,
AgentResponse, MakeAgentsPublicRequest,
PatchAgentRequest)
from litellm.types.proxy.management_endpoints.common_daily_activity import \
SpendAnalyticsPaginatedResponse
router = APIRouter()
@ -69,12 +66,18 @@ async def get_agents(
Returns: List[AgentResponse]
"""
from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import (
AgentRequestHandler,
)
from litellm.proxy.agent_endpoints.agent_registry import \
global_agent_registry
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import \
AgentRequestHandler
try:
from litellm.proxy.proxy_server import \
use_background_agent_health_checks
if not use_background_agent_health_checks:
await global_agent_registry.run_health_check()
returned_agents: List[AgentResponse] = []
# Admin users get all agents
@ -82,7 +85,7 @@ async def get_agents(
user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN
or user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
):
returned_agents = global_agent_registry.get_agent_list()
returned_agents = global_agent_registry.get_healthy_agent_list()
else:
# Get allowed agents from object_permission (key/team level)
allowed_agent_ids = await AgentRequestHandler.get_allowed_agents(
@ -91,10 +94,10 @@ async def get_agents(
# If no restrictions (empty list), return all agents
if len(allowed_agent_ids) == 0:
returned_agents = global_agent_registry.get_agent_list()
returned_agents = global_agent_registry.get_healthy_agent_list()
else:
# Filter agents by allowed IDs
all_agents = global_agent_registry.get_agent_list()
all_agents = global_agent_registry.get_healthy_agent_list()
returned_agents = [
agent for agent in all_agents
if agent.agent_id in allowed_agent_ids
@ -125,9 +128,8 @@ async def get_agents(
#### CRUD ENDPOINTS FOR AGENTS ####
from litellm.proxy.agent_endpoints.agent_registry import (
global_agent_registry as AGENT_REGISTRY,
)
from litellm.proxy.agent_endpoints.agent_registry import \
global_agent_registry as AGENT_REGISTRY
@router.post(
@ -564,9 +566,8 @@ async def make_agent_public(
try:
# Update the public model groups
import litellm
from litellm.proxy.agent_endpoints.agent_registry import (
global_agent_registry as AGENT_REGISTRY,
)
from litellm.proxy.agent_endpoints.agent_registry import \
global_agent_registry as AGENT_REGISTRY
from litellm.proxy.proxy_server import proxy_config
# Check if user has admin permissions
@ -681,9 +682,8 @@ async def make_agents_public(
try:
# Update the public model groups
import litellm
from litellm.proxy.agent_endpoints.agent_registry import (
global_agent_registry as AGENT_REGISTRY,
)
from litellm.proxy.agent_endpoints.agent_registry import \
global_agent_registry as AGENT_REGISTRY
from litellm.proxy.proxy_server import proxy_config
# Load existing config

File diff suppressed because it is too large Load diff