mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
feat(agents/): add health check for agent endpoints before returning
filter out bad endpoints
This commit is contained in:
parent
49ddcba44a
commit
d7c5afa75f
6 changed files with 537 additions and 601 deletions
|
|
@ -257,6 +257,21 @@ LiteLLM follows the [A2A JSON-RPC 2.0 specification](https://github.com/google/A
|
|||
}
|
||||
```
|
||||
|
||||
## Agent Health Checks
|
||||
|
||||
LiteLLM can automatically filter out agents whose backends are unreachable. The `GET /v1/agents` endpoint only returns agents that are currently healthy.
|
||||
|
||||
- **Without background checks (default):** Health checks run inline every time `GET /v1/agents` is called.
|
||||
- **With background checks:** A background loop periodically pings agents so the endpoint responds instantly.
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
background_agent_health_checks: true # run health checks in the background
|
||||
health_check_interval: 300 # interval in seconds (default: 300)
|
||||
```
|
||||
|
||||
For more details, see [Background Agent Health Checks](./proxy/health#background-agent-health-checks).
|
||||
|
||||
## Agent Registry
|
||||
|
||||
Want to create a central registry so your team can discover what agents are available within your company?
|
||||
|
|
|
|||
|
|
@ -132,6 +132,7 @@ general_settings:
|
|||
global_max_parallel_requests: 0 # the max parallel requests allowed on the proxy all up
|
||||
infer_model_from_keys: true
|
||||
background_health_checks: true
|
||||
background_agent_health_checks: true # enable background health checks for A2A agents
|
||||
health_check_interval: 300
|
||||
alerting: ["slack", "email"]
|
||||
alerting_threshold: 0
|
||||
|
|
@ -228,6 +229,7 @@ router_settings:
|
|||
| global_max_parallel_requests | integer | The max parallel requests allowed on the proxy overall |
|
||||
| infer_model_from_keys | boolean | If true, infers the model from the provided keys |
|
||||
| background_health_checks | boolean | If true, enables background health checks. [Doc on health checks](health) |
|
||||
| background_agent_health_checks | boolean | If true, enables background health checks for A2A agents. When disabled, health checks run inline on `GET /v1/agents`. [Doc on health checks](health#background-agent-health-checks) |
|
||||
| health_check_interval | integer | The interval for health checks in seconds [Doc on health checks](health) |
|
||||
| alerting | array of strings | List of alerting methods [Doc on Slack Alerting](alerting) |
|
||||
| alerting_threshold | integer | The threshold for triggering alerts [Doc on Slack Alerting](alerting) |
|
||||
|
|
|
|||
|
|
@ -284,6 +284,30 @@ $ litellm /path/to/config.yaml
|
|||
curl --location 'http://0.0.0.0:4000/health'
|
||||
```
|
||||
|
||||
### Background Agent Health Checks
|
||||
|
||||
You can enable background health checks for A2A agents registered on the proxy. When enabled, a background loop periodically pings each agent's URL and tracks which agents are healthy. The `GET /v1/agents` endpoint then only returns agents that are currently reachable.
|
||||
|
||||
If `background_agent_health_checks` is **not** enabled, the health check runs inline every time `GET /v1/agents` is called.
|
||||
|
||||
**How it works:**
|
||||
- Sends an HTTP `HEAD` request to each agent's `url` (from `agent_card_params`)
|
||||
- Any response with status code < 500 (including `405 Method Not Allowed`) means the agent's server is live
|
||||
- Connection errors or timeouts mark the agent as unhealthy
|
||||
- Agents using the completion bridge (no URL, only `custom_llm_provider`) are always treated as healthy
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
background_agent_health_checks: true # enable background agent health checks
|
||||
health_check_interval: 300 # shared interval for all background health checks (seconds)
|
||||
```
|
||||
|
||||
```bash
|
||||
# Only healthy agents are returned
|
||||
curl -X GET "http://0.0.0.0:4000/v1/agents" \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
### Disable Background Health Checks For Specific Models
|
||||
|
||||
Use this if you want to disable background health checks for specific models.
|
||||
|
|
|
|||
|
|
@ -1,13 +1,12 @@
|
|||
import hashlib
|
||||
import json
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
|
||||
from litellm.proxy.management_helpers.object_permission_utils import (
|
||||
handle_update_object_permission_common,
|
||||
)
|
||||
from litellm.proxy.management_helpers.object_permission_utils import \
|
||||
handle_update_object_permission_common
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
from litellm.types.agents import AgentConfig, AgentResponse, PatchAgentRequest
|
||||
|
||||
|
|
@ -15,6 +14,7 @@ from litellm.types.agents import AgentConfig, AgentResponse, PatchAgentRequest
|
|||
class AgentRegistry:
|
||||
def __init__(self):
|
||||
self.agent_list: List[AgentResponse] = []
|
||||
self.healthy_agent_ids: Optional[Set[str]] = None
|
||||
|
||||
def reset_agent_list(self):
|
||||
self.agent_list = []
|
||||
|
|
@ -43,6 +43,43 @@ class AgentRegistry:
|
|||
public_agent_list.append(agent)
|
||||
return public_agent_list
|
||||
|
||||
async def run_health_check(self) -> None:
|
||||
"""Ping each agent's URL with a HEAD request and update healthy_agent_ids."""
|
||||
from litellm.llms.custom_httpx.http_handler import \
|
||||
get_async_httpx_client
|
||||
from litellm.types.llms.custom_http import httpxSpecialProvider
|
||||
|
||||
client = get_async_httpx_client(llm_provider=httpxSpecialProvider.A2A)
|
||||
healthy: Set[str] = set()
|
||||
for agent in self.agent_list:
|
||||
url = agent.agent_card_params.get("url")
|
||||
if not url:
|
||||
# Completion-bridge agents (no URL) are always treated as healthy
|
||||
if agent.litellm_params and agent.litellm_params.get(
|
||||
"custom_llm_provider"
|
||||
):
|
||||
healthy.add(agent.agent_id)
|
||||
continue
|
||||
try:
|
||||
resp = await client.client.head(url, timeout=10.0)
|
||||
if resp.status_code < 500: # handle 405 Method Not Allowed
|
||||
healthy.add(agent.agent_id)
|
||||
except Exception:
|
||||
pass # skip if unhealthy
|
||||
self.healthy_agent_ids = healthy
|
||||
|
||||
def get_healthy_agent_list(
|
||||
self, agent_names: Optional[List[str]] = None
|
||||
) -> List[AgentResponse]:
|
||||
"""Return only agents whose health check passed.
|
||||
|
||||
If no health data exists yet (healthy_agent_ids is None), returns all agents.
|
||||
"""
|
||||
agents = self.get_agent_list(agent_names=agent_names)
|
||||
if self.healthy_agent_ids is None:
|
||||
return agents
|
||||
return [a for a in agents if a.agent_id in self.healthy_agent_ids]
|
||||
|
||||
def _create_agent_id(self, agent_config: AgentConfig) -> str:
|
||||
return hashlib.sha256(
|
||||
json.dumps(agent_config, sort_keys=True).encode()
|
||||
|
|
@ -149,9 +186,13 @@ class AgentRegistry:
|
|||
created_agent_dict = created_agent.model_dump()
|
||||
if created_agent.object_permission is not None:
|
||||
try:
|
||||
created_agent_dict["object_permission"] = created_agent.object_permission.model_dump()
|
||||
created_agent_dict["object_permission"] = (
|
||||
created_agent.object_permission.model_dump()
|
||||
)
|
||||
except Exception:
|
||||
created_agent_dict["object_permission"] = created_agent.object_permission.dict()
|
||||
created_agent_dict["object_permission"] = (
|
||||
created_agent.object_permission.dict()
|
||||
)
|
||||
return AgentResponse(**created_agent_dict) # type: ignore
|
||||
except Exception as e:
|
||||
raise Exception(f"Error adding agent to DB: {str(e)}")
|
||||
|
|
@ -219,12 +260,10 @@ class AgentRegistry:
|
|||
existing_object_permission_id = existing_agent.get(
|
||||
"object_permission_id"
|
||||
)
|
||||
object_permission_id = (
|
||||
await handle_update_object_permission_common(
|
||||
agent_copy,
|
||||
existing_object_permission_id,
|
||||
prisma_client,
|
||||
)
|
||||
object_permission_id = await handle_update_object_permission_common(
|
||||
agent_copy,
|
||||
existing_object_permission_id,
|
||||
prisma_client,
|
||||
)
|
||||
if object_permission_id is not None:
|
||||
update_data["object_permission_id"] = object_permission_id
|
||||
|
|
@ -241,9 +280,13 @@ class AgentRegistry:
|
|||
patched_agent_dict = patched_agent.model_dump()
|
||||
if patched_agent.object_permission is not None:
|
||||
try:
|
||||
patched_agent_dict["object_permission"] = patched_agent.object_permission.model_dump()
|
||||
patched_agent_dict["object_permission"] = (
|
||||
patched_agent.object_permission.model_dump()
|
||||
)
|
||||
except Exception:
|
||||
patched_agent_dict["object_permission"] = patched_agent.object_permission.dict()
|
||||
patched_agent_dict["object_permission"] = (
|
||||
patched_agent.object_permission.dict()
|
||||
)
|
||||
return AgentResponse(**patched_agent_dict) # type: ignore
|
||||
except Exception as e:
|
||||
raise Exception(f"Error patching agent in DB: {str(e)}")
|
||||
|
|
@ -298,12 +341,10 @@ class AgentRegistry:
|
|||
else None
|
||||
)
|
||||
agent_copy = dict(agent)
|
||||
object_permission_id = (
|
||||
await handle_update_object_permission_common(
|
||||
agent_copy,
|
||||
existing_object_permission_id,
|
||||
prisma_client,
|
||||
)
|
||||
object_permission_id = await handle_update_object_permission_common(
|
||||
agent_copy,
|
||||
existing_object_permission_id,
|
||||
prisma_client,
|
||||
)
|
||||
if object_permission_id is not None:
|
||||
update_data["object_permission_id"] = object_permission_id
|
||||
|
|
@ -318,9 +359,13 @@ class AgentRegistry:
|
|||
updated_agent_dict = updated_agent.model_dump()
|
||||
if updated_agent.object_permission is not None:
|
||||
try:
|
||||
updated_agent_dict["object_permission"] = updated_agent.object_permission.model_dump()
|
||||
updated_agent_dict["object_permission"] = (
|
||||
updated_agent.object_permission.model_dump()
|
||||
)
|
||||
except Exception:
|
||||
updated_agent_dict["object_permission"] = updated_agent.object_permission.dict()
|
||||
updated_agent_dict["object_permission"] = (
|
||||
updated_agent.object_permission.dict()
|
||||
)
|
||||
return AgentResponse(**updated_agent_dict) # type: ignore
|
||||
except Exception as e:
|
||||
raise Exception(f"Error updating agent in DB: {str(e)}")
|
||||
|
|
@ -344,7 +389,9 @@ class AgentRegistry:
|
|||
# object_permission is eagerly loaded via include above
|
||||
if agent.object_permission is not None:
|
||||
try:
|
||||
agent_dict["object_permission"] = agent.object_permission.model_dump()
|
||||
agent_dict["object_permission"] = (
|
||||
agent.object_permission.model_dump()
|
||||
)
|
||||
except Exception:
|
||||
agent_dict["object_permission"] = agent.object_permission.dict()
|
||||
agents.append(agent_dict)
|
||||
|
|
|
|||
|
|
@ -14,19 +14,16 @@ from fastapi import APIRouter, Depends, HTTPException, Request
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy._types import (CommonProxyErrors, LitellmUserRoles,
|
||||
UserAPIKeyAuth)
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import get_daily_activity
|
||||
from litellm.types.agents import (
|
||||
AgentConfig,
|
||||
AgentMakePublicResponse,
|
||||
AgentResponse,
|
||||
MakeAgentsPublicRequest,
|
||||
PatchAgentRequest,
|
||||
)
|
||||
from litellm.types.proxy.management_endpoints.common_daily_activity import (
|
||||
SpendAnalyticsPaginatedResponse,
|
||||
)
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import \
|
||||
get_daily_activity
|
||||
from litellm.types.agents import (AgentConfig, AgentMakePublicResponse,
|
||||
AgentResponse, MakeAgentsPublicRequest,
|
||||
PatchAgentRequest)
|
||||
from litellm.types.proxy.management_endpoints.common_daily_activity import \
|
||||
SpendAnalyticsPaginatedResponse
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
|
@ -69,12 +66,18 @@ async def get_agents(
|
|||
Returns: List[AgentResponse]
|
||||
|
||||
"""
|
||||
from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
|
||||
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import (
|
||||
AgentRequestHandler,
|
||||
)
|
||||
from litellm.proxy.agent_endpoints.agent_registry import \
|
||||
global_agent_registry
|
||||
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import \
|
||||
AgentRequestHandler
|
||||
|
||||
try:
|
||||
from litellm.proxy.proxy_server import \
|
||||
use_background_agent_health_checks
|
||||
|
||||
if not use_background_agent_health_checks:
|
||||
await global_agent_registry.run_health_check()
|
||||
|
||||
returned_agents: List[AgentResponse] = []
|
||||
|
||||
# Admin users get all agents
|
||||
|
|
@ -82,7 +85,7 @@ async def get_agents(
|
|||
user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN
|
||||
or user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
|
||||
):
|
||||
returned_agents = global_agent_registry.get_agent_list()
|
||||
returned_agents = global_agent_registry.get_healthy_agent_list()
|
||||
else:
|
||||
# Get allowed agents from object_permission (key/team level)
|
||||
allowed_agent_ids = await AgentRequestHandler.get_allowed_agents(
|
||||
|
|
@ -91,10 +94,10 @@ async def get_agents(
|
|||
|
||||
# If no restrictions (empty list), return all agents
|
||||
if len(allowed_agent_ids) == 0:
|
||||
returned_agents = global_agent_registry.get_agent_list()
|
||||
returned_agents = global_agent_registry.get_healthy_agent_list()
|
||||
else:
|
||||
# Filter agents by allowed IDs
|
||||
all_agents = global_agent_registry.get_agent_list()
|
||||
all_agents = global_agent_registry.get_healthy_agent_list()
|
||||
returned_agents = [
|
||||
agent for agent in all_agents
|
||||
if agent.agent_id in allowed_agent_ids
|
||||
|
|
@ -125,9 +128,8 @@ async def get_agents(
|
|||
|
||||
#### CRUD ENDPOINTS FOR AGENTS ####
|
||||
|
||||
from litellm.proxy.agent_endpoints.agent_registry import (
|
||||
global_agent_registry as AGENT_REGISTRY,
|
||||
)
|
||||
from litellm.proxy.agent_endpoints.agent_registry import \
|
||||
global_agent_registry as AGENT_REGISTRY
|
||||
|
||||
|
||||
@router.post(
|
||||
|
|
@ -564,9 +566,8 @@ async def make_agent_public(
|
|||
try:
|
||||
# Update the public model groups
|
||||
import litellm
|
||||
from litellm.proxy.agent_endpoints.agent_registry import (
|
||||
global_agent_registry as AGENT_REGISTRY,
|
||||
)
|
||||
from litellm.proxy.agent_endpoints.agent_registry import \
|
||||
global_agent_registry as AGENT_REGISTRY
|
||||
from litellm.proxy.proxy_server import proxy_config
|
||||
|
||||
# Check if user has admin permissions
|
||||
|
|
@ -681,9 +682,8 @@ async def make_agents_public(
|
|||
try:
|
||||
# Update the public model groups
|
||||
import litellm
|
||||
from litellm.proxy.agent_endpoints.agent_registry import (
|
||||
global_agent_registry as AGENT_REGISTRY,
|
||||
)
|
||||
from litellm.proxy.agent_endpoints.agent_registry import \
|
||||
global_agent_registry as AGENT_REGISTRY
|
||||
from litellm.proxy.proxy_server import proxy_config
|
||||
|
||||
# Load existing config
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
Loading…
Add table
Reference in a new issue