fix(proxy): expand access groups in /health scoping and allowlist health display fields

/health filtered deployments by the key's literal `models` entries, so a key or UI user
scoped to a model access group got 0/0 on both the live and the background-cache path,
plus a misleading "missing model_info.id" warning on the cached one. The endpoint now
resolves key sentinels and expands access groups the way auth does.

Health entries used to copy every litellm_params field, so a deployment parameter that
is not JSON-safe (a nested mapping keyed by a tuple) 500d the endpoint for every caller
and admins saw internal settings nobody asked for. Entries now keep only an explicit
allowlist of JSON-safe diagnostic fields; api_base and api_version stay admin-only and
credentials stay out for everyone.

Fixes #28206
This commit is contained in:
mateo-berri 2026-09-03 17:24:42 -07:00
parent 170a1eeb06
commit e75dd56631
3 changed files with 342 additions and 55 deletions

View file

@ -32,32 +32,35 @@ from litellm.router_utils.auto_router_model_naming import (
strategy_router_dependencies,
)
ILLEGAL_DISPLAY_PARAMS: Final = [
"messages",
"api_key",
"prompt",
"input",
"client_secret",
"azure_ad_token",
"azure_username",
"azure_password",
"vertex_credentials",
"vertex_ai_credentials",
"aws_access_key_id",
"aws_secret_access_key",
"aws_session_token",
"aws_web_identity_token",
"extra_headers",
"headers",
"exception", # internal; not JSON-serializable, never for display
"litellm_metadata", # internal tracking metadata with auth objects; not for display
]
# Provider routing fields. Allowed for proxy admins so they can see which
# region/version a deployment is checking; gated at the endpoint layer for
# non-admin callers (see _strip_admin_only_fields_from_health_result).
ADMIN_ONLY_HEALTH_DISPLAY_PARAMS: Final = ("api_base", "api_version")
MINIMAL_DISPLAY_PARAMS: Final = ["model", "mode_error"]
MINIMAL_DISPLAY_PARAMS: Final = frozenset({"model", "mode_error"})
HEALTH_DISPLAY_PARAMS: Final = (
MINIMAL_DISPLAY_PARAMS
| frozenset(ADMIN_ONLY_HEALTH_DISPLAY_PARAMS)
| frozenset(
{
"custom_llm_provider",
"mode",
"base_model",
"aws_region_name",
"region_name",
"vertex_project",
"vertex_location",
"tpm",
"rpm",
"error",
"raw_request_typed_dict",
"x-ratelimit-remaining-requests",
"x-ratelimit-remaining-tokens",
"x-ms-region",
}
)
)
# Modes whose health-check probe is a chat-style completion call and
# therefore accept `max_tokens`. Other modes (embedding, image_generation,
@ -143,14 +146,10 @@ def _get_random_llm_message():
def _clean_endpoint_data(endpoint_data: dict, details: bool | None = True):
"""
Clean the endpoint data for display to users.
Keep only the explicitly approved, JSON-safe diagnostic fields for display to users.
"""
endpoint_data.pop("litellm_logging_obj", None)
return (
{k: v for k, v in endpoint_data.items() if k not in ILLEGAL_DISPLAY_PARAMS}
if details is not False
else {k: v for k, v in endpoint_data.items() if k in MINIMAL_DISPLAY_PARAMS}
)
displayed: Final = HEALTH_DISPLAY_PARAMS if details is not False else MINIMAL_DISPLAY_PARAMS
return {k: v for k, v in endpoint_data.items() if k in displayed}
def health_check_filter_kwargs_from_general_settings(

View file

@ -36,9 +36,13 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
WebhookEvent,
)
from litellm.proxy.auth.auth_checks import (
_resolve_key_models_for_auth_check, # pyright: ignore[reportPrivateUsage] # the auth layer's sentinel resolution, reused so /health scopes exactly like a request
)
from litellm.proxy.auth.auth_utils import (
_BANNED_REQUEST_BODY_PARAMS, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the request-body check
)
from litellm.proxy.auth.model_checks import get_key_models
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler
from litellm.proxy.db.proxy_worker_heartbeat import count_live_proxy_workers
@ -54,6 +58,7 @@ from litellm.proxy.middleware.in_flight_requests_middleware import (
get_in_flight_requests,
)
from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager
from litellm.router import Router
from litellm.router_utils.clientside_credential_handler import (
_ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the router path
clientside_credential_keys,
@ -854,6 +859,24 @@ def _strip_admin_only_fields_from_health_result(result: dict) -> dict:
return out
def _health_accessible_model_names(
user_api_key_dict: UserAPIKeyAuth, llm_router: Router | None
) -> frozenset[str] | None:
"""Model names the caller may health-check, or None when the key is unrestricted."""
granted_models: Final = _resolve_key_models_for_auth_check(user_api_key_dict)
if not granted_models or SpecialModelNames.all_proxy_models.value in granted_models:
return None
if llm_router is None:
return frozenset(granted_models)
return frozenset(
get_key_models(
user_api_key_dict=user_api_key_dict,
proxy_model_list=llm_router.get_model_names(team_id=user_api_key_dict.team_id),
model_access_groups=llm_router.get_model_access_groups(),
)
)
def _resolve_targeted_model_ids(model_list: list, model: str | None, model_id: str | None) -> set | None:
"""
Resolve a ``/health`` ``model`` / ``model_id`` query param to the set of
@ -1080,32 +1103,11 @@ async def health_endpoint(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail={"error": "Model list not initialized"},
)
_llm_model_list = copy.deepcopy(llm_model_list)
### FILTER MODELS FOR ONLY THOSE USER HAS ACCESS TO ###
# Live path: scope by model_name (every deployment has one).
# Cache path: scope by model_id (the cache is keyed on model_id).
# Consequence: a deployment whose model_name the caller can access
# but which lacks model_info.id will appear in the live /health
# response but NOT in the background-cache /health response. This is
# surfaced via the "warnings" field below so operators can fix the
# missing model_info.id rather than guess at the discrepancy.
# Keys granted SpecialModelNames.all_proxy_models carry the literal
# "all-proxy-models" entry, which matches no real model_name; treat
# them as unrestricted instead of filtering the list down to nothing.
# Keys granted SpecialModelNames.all_team_models inherit the parent
# team's allowlist (same semantics as get_key_models in
# model_checks.py). Without a team_id the sentinel cannot resolve and
# stays in the list, matching nothing; denied rather than
# unrestricted, mirroring _resolve_key_models_for_auth_check.
accessible_models = list(user_api_key_dict.models)
if SpecialModelNames.all_team_models.value in accessible_models and user_api_key_dict.team_id is not None:
accessible_models = list(user_api_key_dict.team_models)
restrict_to_allowed_models: Final = (
len(accessible_models) > 0 and SpecialModelNames.all_proxy_models.value not in accessible_models
)
if restrict_to_allowed_models:
allowed_models: Final = set(accessible_models)
_llm_model_list = [m for m in _llm_model_list if m.get("model_name") in allowed_models]
allowed_models: Final = _health_accessible_model_names(user_api_key_dict, llm_router)
restrict_to_allowed_models: Final = allowed_models is not None
_llm_model_list: Final = [
m for m in copy.deepcopy(llm_model_list) if allowed_models is None or m.get("model_name") in allowed_models
]
if use_background_health_checks:
# The cached background result covers every model. When the
# caller targets a specific model/model_id we have to narrow the
@ -1125,7 +1127,7 @@ async def health_endpoint(
# intersection of "targeted" and "allowed."
filter_ids: Final = targeted_ids if targeted_ids is not None else allowed_model_ids
filtered: Final = _filter_health_check_results_by_model_ids(health_check_results, filter_ids)
if targeted_ids is None and not allowed_model_ids:
if targeted_ids is None and _llm_model_list and not allowed_model_ids:
# Caller has accessible model_names but none of the
# matching deployments expose a model_info.id, so the
# cache filter (which keys on model_id) drops every

View file

@ -1,6 +1,8 @@
import asyncio
import json
import time
from collections.abc import Iterator, Mapping, Sequence
from contextlib import contextmanager
from datetime import datetime, timedelta
from types import SimpleNamespace
from unittest.mock import AsyncMock, MagicMock, patch
@ -1650,6 +1652,209 @@ async def test_health_endpoint_resolves_all_team_models_to_team_allowlist():
assert returned_names == {"model-b"}, f"all-team-models key should health-check the team's models: {returned_names}"
class _AccessGroupRouter:
"""Router stand-in exposing only the two lookups /health uses to expand a key's model grants."""
def __init__(self, access_groups: dict[str, list[str]], model_names: list[str]) -> None:
self._access_groups = access_groups
self._model_names = model_names
def get_model_access_groups(self, model_name=None, model_access_group=None, team_id=None):
return self._access_groups
def get_model_names(self, team_id=None):
return self._model_names
_ACCESS_GROUP_MODEL_LIST = [
{
"model_name": "bedrock-nova",
"litellm_params": {"model": "bedrock/us.amazon.nova-2-lite-v1:0"},
"model_info": {"id": "id-bedrock", "access_groups": ["bedrock-group"]},
},
{
"model_name": "gpt-5.4-mini",
"litellm_params": {"model": "openai/gpt-5.4-mini"},
"model_info": {"id": "id-openai"},
},
]
_ACCESS_GROUP_ROUTER = _AccessGroupRouter({"bedrock-group": ["bedrock-nova"]}, ["bedrock-nova", "gpt-5.4-mini"])
_ACCESS_GROUP_CACHED_RESULTS = {
"healthy_endpoints": [
{"model": "bedrock/us.amazon.nova-2-lite-v1:0", "model_id": "id-bedrock"},
{"model": "openai/gpt-5.4-mini", "model_id": "id-openai"},
],
"unhealthy_endpoints": [],
"healthy_count": 2,
"unhealthy_count": 0,
}
@contextmanager
def _proxy_health_globals(
llm_model_list: Sequence[Mapping[str, object]],
llm_router: object,
use_background_health_checks: bool = False,
health_check_results: Mapping[str, object] | None = None,
) -> Iterator[None]:
with (
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.llm_model_list", list(llm_model_list)
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.llm_router", llm_router
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.prisma_client", None
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.use_background_health_checks", use_background_health_checks
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.user_model", None
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.health_check_results", dict(health_check_results or {})
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.health_check_details", True
),
patch( # test-quality-ok: proxy module global, no injection seam
"litellm.proxy.proxy_server.health_check_concurrency", 1
),
):
yield
@pytest.mark.asyncio
async def test_health_endpoint_expands_access_group_on_live_path():
"""
LIT-6907 / gh-28206: a key granted a model access group carries the group
name in user_api_key_dict.models. Matching it as a literal model_name
filtered every deployment out and /health answered 0/0 for a model the
same key could call.
"""
from fastapi import Response
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.health_endpoints._health_endpoints import health_endpoint
captured: dict = {}
async def fake_perform(**kwargs):
captured["model_list"] = kwargs["model_list"]
return {"healthy_endpoints": [], "unhealthy_endpoints": [], "healthy_count": 0, "unhealthy_count": 0}
with (
_proxy_health_globals(_ACCESS_GROUP_MODEL_LIST, _ACCESS_GROUP_ROUTER),
patch( # test-quality-ok: the model list handed to the probe is the assertion; no injection seam
"litellm.proxy.health_endpoints._health_endpoints._perform_health_check_and_save",
side_effect=fake_perform,
),
):
await health_endpoint(
response=Response(),
user_api_key_dict=UserAPIKeyAuth(api_key="hashed-test-key", models=["bedrock-group"]),
)
assert [m["model_name"] for m in captured["model_list"]] == ["bedrock-nova"]
@pytest.mark.asyncio
async def test_health_endpoint_expands_access_group_on_background_cache_path():
"""
LIT-6907: the background-cache path scoped the cached entries through the
same literal model_name match, so an access-group key got an empty result
plus a warning blaming missing model_info.id.
"""
from fastapi import Response
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.health_endpoints._health_endpoints import health_endpoint
with _proxy_health_globals(
_ACCESS_GROUP_MODEL_LIST,
_ACCESS_GROUP_ROUTER,
use_background_health_checks=True,
health_check_results=_ACCESS_GROUP_CACHED_RESULTS,
):
result = await health_endpoint(
response=Response(),
user_api_key_dict=UserAPIKeyAuth(api_key="hashed-test-key", models=["bedrock-group"]),
model=None,
model_id=None,
)
assert [e["model_id"] for e in result["healthy_endpoints"]] == ["id-bedrock"]
assert result["healthy_count"] == 1
assert "warnings" not in result
@pytest.mark.asyncio
async def test_health_endpoint_treats_no_team_all_team_models_as_unrestricted():
"""
A key granted "all-team-models" without a team resolves to an empty
allowlist in the auth layer, which means unrestricted. /health used to
keep the unresolved sentinel and filter every deployment out instead.
"""
from fastapi import Response
from litellm.proxy._types import SpecialModelNames, UserAPIKeyAuth
from litellm.proxy.health_endpoints._health_endpoints import health_endpoint
captured: dict = {}
async def fake_perform(**kwargs):
captured["model_list"] = kwargs["model_list"]
return {"healthy_endpoints": [], "unhealthy_endpoints": [], "healthy_count": 0, "unhealthy_count": 0}
with (
_proxy_health_globals(_ACCESS_GROUP_MODEL_LIST, None),
patch( # test-quality-ok: the model list handed to the probe is the assertion; no injection seam
"litellm.proxy.health_endpoints._health_endpoints._perform_health_check_and_save",
side_effect=fake_perform,
),
):
await health_endpoint(
response=Response(),
user_api_key_dict=UserAPIKeyAuth(
api_key="hashed-test-key", models=[SpecialModelNames.all_team_models.value], team_id=None
),
)
assert {m["model_name"] for m in captured["model_list"]} == {"bedrock-nova", "gpt-5.4-mini"}
@pytest.mark.asyncio
async def test_health_endpoint_omits_model_id_warning_when_no_deployment_matches():
"""
The missing-model_info.id warning is only true when a matching deployment
exists without an id. A key whose grants match no deployment at all gets a
plain empty result, not advice to populate ids that are already there.
"""
from fastapi import Response
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.health_endpoints._health_endpoints import health_endpoint
with _proxy_health_globals(
_ACCESS_GROUP_MODEL_LIST,
_ACCESS_GROUP_ROUTER,
use_background_health_checks=True,
health_check_results=_ACCESS_GROUP_CACHED_RESULTS,
):
result = await health_endpoint(
response=Response(),
user_api_key_dict=UserAPIKeyAuth(api_key="hashed-test-key", models=["no-such-model"]),
model=None,
model_id=None,
)
assert result["healthy_count"] == 0
assert result["unhealthy_count"] == 0
assert "warnings" not in result
@pytest.mark.asyncio
async def test_health_endpoint_filters_background_cache_by_user_access():
"""
@ -2571,6 +2776,87 @@ def test_clean_endpoint_data_never_displays_credential_fields(credential_field,
assert canary not in str(cleaned)
def test_clean_endpoint_data_keeps_only_json_safe_diagnostics():
"""
LIT-6907: _clean_endpoint_data used to copy every litellm_param not on a
deny list, so a nested mapping keyed by a tuple reached jsonable_encoder
and 500'd /health. Only the explicit allowlist survives now.
"""
from fastapi.encoders import jsonable_encoder
from litellm.proxy.health_check import _clean_endpoint_data
cleaned = _clean_endpoint_data(
{
"model": "bedrock/us.amazon.nova-2-lite-v1:0",
"custom_llm_provider": "bedrock",
"aws_region_name": "us-east-1",
"metadata": {("us-east-1", "primary"): "canary-nested-mapping"},
"allow_client_keepalive_override": False,
"api_key": "CANARY-API-KEY",
"x-ratelimit-remaining-requests": 99,
"raw_request_typed_dict": {"raw_request_api_base": "https://example.test"},
},
details=True,
)
assert cleaned == {
"model": "bedrock/us.amazon.nova-2-lite-v1:0",
"custom_llm_provider": "bedrock",
"aws_region_name": "us-east-1",
"x-ratelimit-remaining-requests": 99,
"raw_request_typed_dict": {"raw_request_api_base": "https://example.test"},
}
assert jsonable_encoder(cleaned) == cleaned
@pytest.mark.asyncio
async def test_health_endpoint_result_survives_non_json_safe_deployment_params():
"""
LIT-6907: the full /health path with a deployment carrying a tuple-keyed
nested mapping must produce a response FastAPI can encode, with the
approved diagnostics intact and the offending param absent.
"""
from fastapi import Response
from fastapi.encoders import jsonable_encoder
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.health_endpoints._health_endpoints import health_endpoint
model_list = [
{
"model_name": "bedrock-nova",
"litellm_params": {
"model": "bedrock/us.amazon.nova-2-lite-v1:0",
"aws_region_name": "us-east-1",
"aws_access_key_id": "CANARY-ACCESS-KEY",
"metadata": {("us-east-1", "primary"): "canary-nested-mapping"},
},
"model_info": {"id": "id-bedrock"},
}
]
with (
_proxy_health_globals(model_list, None),
patch( # test-quality-ok: the provider probe is faked; the assertion is the response shaping after it
"litellm.ahealth_check", AsyncMock(return_value={"x-ratelimit-remaining-requests": 99})
),
):
result = await health_endpoint(
response=Response(),
user_api_key_dict=UserAPIKeyAuth(api_key="hashed-admin-key", user_role=LitellmUserRoles.PROXY_ADMIN),
)
encoded = jsonable_encoder(result)
assert encoded["healthy_count"] == 1
entry = encoded["healthy_endpoints"][0]
assert entry["model_id"] == "id-bedrock"
assert entry["aws_region_name"] == "us-east-1"
assert entry["x-ratelimit-remaining-requests"] == 99
assert "metadata" not in entry
assert "CANARY" not in str(encoded)
class TestConfigBaseForHealthCheck:
"""A request that sets its own connection fields gets a base without the
configuration's credentials; anything it leaves unset still comes from