From 232581c6c0087a801f2c49b9b07f15f5c2958727 Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 07:03:49 +0530 Subject: [PATCH 1/6] fix(proxy): let internal users test the connection of models they can call /health/test_connection authorized every request through ModelManagementAuthChecks.can_user_make_model_call, the check used for adding, updating and deleting models. For a model without a team_id that means proxy admins only, so the "Test Connection" button the UI shows to every user failed with 403 for internal users. Testing a connection is a read against a configured deployment, so use the same permission as calling the model: after the management check denies a non-team model, allow the request when the deployment was resolved from the router, the request does not override any connection field (a request that sets api_base, api_key, ... describes a different endpoint and stays a management operation), and the caller's key and user may call the deployment's model_name. Fixes #40265 --- .../health_endpoints/_health_endpoints.py | 74 +++++++++- .../health_endpoints/test_health_endpoints.py | 126 ++++++++++++++++++ 2 files changed, 196 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index db6ec754c6e..f6174f62ab4 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -1997,6 +1997,69 @@ async def health_liveliness_options(): return Response(headers=response_headers, status_code=200) +async def _authorize_test_connection( + *, + model_params: Any, + user_api_key_dict: UserAPIKeyAuth, + prisma_client: Any, + premium_user: bool, + llm_router: Any, + configured_model_name: str | None, + request_litellm_params: Mapping[str, object], +) -> None: + """Decide whether the caller may probe this model. + + Proxy admins and team admins may probe any model they manage, as before. + Any other user may probe a configured model they are allowed to call, but + only as configured: a request that sets its own connection fields describes + a different endpoint, and probing that stays a management operation. + """ + from litellm.proxy.auth.auth_checks import ( + can_key_call_model, + can_user_call_model, + get_user_object, + ) + from litellm.proxy.management_endpoints.model_management_endpoints import ( + ModelManagementAuthChecks, + ) + from litellm.proxy.proxy_server import llm_model_list, user_api_key_cache + + try: + await ModelManagementAuthChecks.can_user_make_model_call( + model_params=model_params, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, + premium_user=premium_user, + ) + return + except HTTPException as management_denial: + if management_denial.status_code != 403: + raise + if configured_model_name is None or any(field in request_litellm_params for field in _CONFIG_CONNECTION_FIELDS): + raise + + try: + await can_key_call_model( + model=configured_model_name, + llm_model_list=llm_model_list, + valid_token=user_api_key_dict, + llm_router=llm_router, + ) + user_object = await get_user_object( + user_id=user_api_key_dict.user_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + user_id_upsert=False, + ) + await can_user_call_model( + model=configured_model_name, + llm_router=llm_router, + user_object=user_object, + ) + except ProxyException as e: + raise HTTPException(status_code=403, detail={"error": str(e.message)}) from e + + @router.post( "/health/test_connection", tags=["health"], @@ -2083,9 +2146,6 @@ async def test_model_connection( dict: A dictionary containing the health check result with either success information or error details. """ from litellm.proxy._types import CommonProxyErrors - from litellm.proxy.management_endpoints.model_management_endpoints import ( - ModelManagementAuthChecks, - ) from litellm.proxy.proxy_server import ( general_settings, llm_router, @@ -2115,6 +2175,7 @@ async def test_model_connection( # This gets the litellm_params from proxy config (with resolved env vars) config_litellm_params: dict = {} loaded_model_info: dict | None = None + configured_model_name: str | None = None if llm_router is not None: # Prefer disambiguation by deployment id (`model_info.id`) when # the caller supplies it. This is required when multiple @@ -2134,6 +2195,7 @@ async def test_model_connection( if deployment_by_id is not None: config_litellm_params = deployment_by_id.litellm_params.model_dump(exclude_none=True) loaded_model_info = deployment_by_id.model_info.model_dump(exclude_none=True) + configured_model_name = deployment_by_id.model_name elif model_name: # Fall back to model_name lookup for callers (e.g. the # "Add Model" wizard, or curl) that don't supply an id. @@ -2156,6 +2218,7 @@ async def test_model_connection( # variables from proxy config. config_litellm_params = dict(deployments[0].get("litellm_params", {})) loaded_model_info = dict(deployments[0].get("model_info") or {}) + configured_model_name = deployments[0].get("model_name") except Exception as e: verbose_proxy_logger.debug( "Could not find model %s in router: %s. Proceeding with request params only.", model_name, e @@ -2178,7 +2241,7 @@ async def test_model_connection( ) ## Auth check, on the final probe params so health_check_params cannot retarget it afterwards - await ModelManagementAuthChecks.can_user_make_model_call( + await _authorize_test_connection( model_params=Deployment( model_name="test_model", litellm_params=LiteLLM_Params(**litellm_params), @@ -2187,6 +2250,9 @@ async def test_model_connection( user_api_key_dict=user_api_key_dict, prisma_client=prisma_client, premium_user=premium_user, + llm_router=llm_router, + configured_model_name=configured_model_name, + request_litellm_params=request_litellm_params, ) mode = mode or litellm_params.pop("mode", None) diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index 761cd0685f2..ecf4523fb45 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -4179,3 +4179,129 @@ async def test_health_services_endpoint_pointfive_blocks_non_admin(monkeypatch, assert str(raised.value.code) == "403" logger_class.assert_not_called() + + +def _configured_non_team_deployment(): + from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo + + return Deployment( + model_name="gpt-4o", + litellm_params=LiteLLM_Params( + model="openai/gpt-4o", + api_base="https://configured.invalid/v1", + api_key="CONFIGURED-API-KEY", + ), + model_info=ModelInfo(id="non-team-deployment-id"), + ) + + +def _internal_user(): + return UserAPIKeyAuth( + token="internal-user-token", + user_id="internal-user", + user_role=LitellmUserRoles.INTERNAL_USER, + ) + + +@pytest.mark.asyncio +async def test_test_model_connection_allows_internal_user_to_probe_configured_model_they_can_call(): + """ + An internal user may test a configured, non-team model they are allowed + to call, and the probe runs with the configured credentials. + """ + mock_router = MagicMock() + mock_router.get_deployment.return_value = _configured_non_team_deployment() + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, + patch("litellm.proxy.auth.auth_checks.get_user_object", AsyncMock(return_value=None)), + patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)) as user_check, + patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})) as health_check, + ): + result = await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "openai/gpt-4o"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert result["status"] == "success" + assert key_check.await_args.kwargs["model"] == "gpt-4o" + assert user_check.await_args.kwargs["model"] == "gpt-4o" + assert health_check.await_args.kwargs["model_params"]["api_key"] == "CONFIGURED-API-KEY" + + +@pytest.mark.asyncio +async def test_test_model_connection_denies_internal_user_without_model_access(): + from fastapi import HTTPException + + from litellm.proxy._types import ProxyErrorTypes, ProxyException + + mock_router = MagicMock() + mock_router.get_deployment.return_value = _configured_non_team_deployment() + denied = ProxyException( + message="Key not allowed to access model. Tried to access gpt-4o", + type=ProxyErrorTypes.key_model_access_denied, + param="model", + code=403, + ) + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(side_effect=denied)), + patch("litellm.ahealth_check", AsyncMock()) as health_check, + pytest.raises(HTTPException) as exc_info, + ): + await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "openai/gpt-4o"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert exc_info.value.status_code == 403 + assert "not allowed to access model" in exc_info.value.detail["error"] + health_check.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_test_model_connection_keeps_connection_overrides_admin_only(): + from fastapi import HTTPException + + """ + A request that sets its own connection fields describes a different + endpoint than the configured one; probing that stays a management + operation, so an internal user is denied before any access check runs. + """ + mock_router = MagicMock() + mock_router.get_deployment.return_value = _configured_non_team_deployment() + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, + patch("litellm.ahealth_check", AsyncMock()) as health_check, + pytest.raises(HTTPException) as exc_info, + ): + await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={ + "model": "openai/gpt-4o", + "api_base": "https://somewhere-else.invalid/v1", + }, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert exc_info.value.status_code == 403 + key_check.assert_not_awaited() + health_check.assert_not_awaited() From 9f9fc047293e3d98abced60c534e8c71eacb5dc3 Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 15:36:48 +0530 Subject: [PATCH 2/6] fix(proxy): tighten the non-admin test_connection path Address review findings on the non-admin fallback: - Only fall back for deployments without a team_id; team deployments keep their team-admin policy instead of falling through to key/user checks. - Replace the connection-field denylist with an equality rule: a non-admin request may not differ from the configured deployment in any litellm_params value (model, provider, endpoint, credentials, ...), so the access check and the probe always target the same deployment. - Treat a missing user record (UserNotFoundError) as "no user-level restrictions" instead of surfacing a 500. - Limit the response for callers admitted through this path to the model and the probe outcome, hiding api_base/api_version and the rest of the deployment's configuration. - Type the helper's parameters (Deployment, PrismaClient, Router). --- .../health_endpoints/_health_endpoints.py | 64 +++++++--- .../health_endpoints/test_health_endpoints.py | 116 ++++++++++++++++++ 2 files changed, 162 insertions(+), 18 deletions(-) diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index f6174f62ab4..ffb48a460b1 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -8,7 +8,7 @@ import time import traceback from collections.abc import Iterable, Mapping from datetime import datetime, timedelta, timezone -from typing import Any, Final, Literal, TypedDict, cast +from typing import TYPE_CHECKING, Any, Final, Literal, TypedDict, cast import fastapi from fastapi import APIRouter, Depends, HTTPException, Request, Response, status @@ -64,6 +64,10 @@ from litellm.proxy.middleware.in_flight_requests_middleware import ( ) from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager from litellm.router import Router + +if TYPE_CHECKING: + from litellm.proxy.utils import PrismaClient + from litellm.types.router import Deployment from litellm.router_utils.clientside_credential_handler import ( _ADMIN_CONFIG_FIELDS_TO_CLEAR_ON_BASE_OVERRIDE, # pyright: ignore[reportPrivateUsage] # one canonical list, shared with the router path clientside_credential_keys, @@ -1997,24 +2001,36 @@ async def health_liveliness_options(): return Response(headers=response_headers, status_code=200) +# What a caller admitted through the non-admin path gets to see of the probe: +# the model and the outcome, none of the deployment's routing configuration. +_NON_ADMIN_TEST_CONNECTION_RESULT_KEYS: Final[frozenset[str]] = frozenset(("model", "error", "mode_error")) + + async def _authorize_test_connection( *, - model_params: Any, + model_params: "Deployment", user_api_key_dict: UserAPIKeyAuth, - prisma_client: Any, + prisma_client: "PrismaClient", premium_user: bool, - llm_router: Any, + llm_router: "Router | None", configured_model_name: str | None, + configured_litellm_params: Mapping[str, object], request_litellm_params: Mapping[str, object], -) -> None: +) -> bool: """Decide whether the caller may probe this model. Proxy admins and team admins may probe any model they manage, as before. - Any other user may probe a configured model they are allowed to call, but - only as configured: a request that sets its own connection fields describes - a different endpoint, and probing that stays a management operation. + Any other user may probe a configured deployment that has no team, exactly + as configured, when their key and user are allowed to call its model: + a request value that differs from the configuration (model, provider, + endpoint, credentials, ...) describes a different probe, and that stays a + management operation. Team deployments keep their team-admin policy. + + Returns True when the caller was admitted through that non-admin path, so + the response can be limited to the outcome of the probe. """ from litellm.proxy.auth.auth_checks import ( + UserNotFoundError, can_key_call_model, can_user_call_model, get_user_object, @@ -2031,11 +2047,16 @@ async def _authorize_test_connection( prisma_client=prisma_client, premium_user=premium_user, ) - return + return False except HTTPException as management_denial: if management_denial.status_code != 403: raise - if configured_model_name is None or any(field in request_litellm_params for field in _CONFIG_CONNECTION_FIELDS): + if configured_model_name is None or getattr(model_params.model_info, "team_id", None) is not None: + raise + if any( + key != "mode" and configured_litellm_params.get(key) != value + for key, value in request_litellm_params.items() + ): raise try: @@ -2045,12 +2066,15 @@ async def _authorize_test_connection( valid_token=user_api_key_dict, llm_router=llm_router, ) - user_object = await get_user_object( - user_id=user_api_key_dict.user_id, - prisma_client=prisma_client, - user_api_key_cache=user_api_key_cache, - user_id_upsert=False, - ) + try: + user_object = await get_user_object( + user_id=user_api_key_dict.user_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + user_id_upsert=False, + ) + except UserNotFoundError: + user_object = None await can_user_call_model( model=configured_model_name, llm_router=llm_router, @@ -2058,6 +2082,7 @@ async def _authorize_test_connection( ) except ProxyException as e: raise HTTPException(status_code=403, detail={"error": str(e.message)}) from e + return True @router.post( @@ -2241,7 +2266,7 @@ async def test_model_connection( ) ## Auth check, on the final probe params so health_check_params cannot retarget it afterwards - await _authorize_test_connection( + admitted_as_caller: Final = await _authorize_test_connection( model_params=Deployment( model_name="test_model", litellm_params=LiteLLM_Params(**litellm_params), @@ -2252,6 +2277,7 @@ async def test_model_connection( premium_user=premium_user, llm_router=llm_router, configured_model_name=configured_model_name, + configured_litellm_params=config_litellm_params, request_litellm_params=request_litellm_params, ) mode = mode or litellm_params.pop("mode", None) @@ -2267,7 +2293,9 @@ async def test_model_connection( ) # Clean the result for display - cleaned_result: Final = _clean_endpoint_data({**litellm_params, **result}, details=True) + cleaned_result = _clean_endpoint_data({**litellm_params, **result}, details=True) + if admitted_as_caller: + cleaned_result = {k: v for k, v in cleaned_result.items() if k in _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS} return { "status": "error" if "error" in result else "success", diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index ecf4523fb45..543c4e2453d 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -4233,6 +4233,9 @@ async def test_test_model_connection_allows_internal_user_to_probe_configured_mo assert key_check.await_args.kwargs["model"] == "gpt-4o" assert user_check.await_args.kwargs["model"] == "gpt-4o" assert health_check.await_args.kwargs["model_params"]["api_key"] == "CONFIGURED-API-KEY" + # the caller learns the outcome, not the deployment's routing configuration + assert "api_base" not in result["result"] + assert "api_key" not in result["result"] @pytest.mark.asyncio @@ -4305,3 +4308,116 @@ async def test_test_model_connection_keeps_connection_overrides_admin_only(): assert exc_info.value.status_code == 403 key_check.assert_not_awaited() health_check.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_test_model_connection_keeps_model_overrides_admin_only(): + """ + The probe runs whatever `litellm_params.model` the request sends, so a + non-admin may only send the configured deployment's own model: the access + check and the probe must target the same model. + """ + from fastapi import HTTPException + + mock_router = MagicMock() + mock_router.get_deployment.return_value = _configured_non_team_deployment() + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, + patch("litellm.ahealth_check", AsyncMock()) as health_check, + pytest.raises(HTTPException) as exc_info, + ): + await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "openai/gpt-4o-mini"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert exc_info.value.status_code == 403 + key_check.assert_not_awaited() + health_check.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_admins(): + """ + A team deployment keeps its team-admin policy: a non-admin whose key can + call the same model name must not reach the non-admin path with the + team's configured credentials. + """ + from fastapi import HTTPException + + from litellm.proxy._types import LiteLLM_TeamTable + from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo + + team_deployment = Deployment( + model_name="gpt-4o", + litellm_params=LiteLLM_Params(model="openai/gpt-4o", api_key="TEAM-A-API-KEY"), + model_info=ModelInfo(id="team-a-deployment-id", team_id="team-a"), + ) + mock_router = MagicMock() + mock_router.get_deployment.return_value = team_deployment + team_row = SimpleNamespace( + model_dump=lambda: LiteLLM_TeamTable(team_id="team-a", members_with_roles=[]).model_dump() + ) + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.management_endpoints.model_management_endpoints.TeamRepository") as MockTeamRepo, + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, + patch("litellm.ahealth_check", AsyncMock()) as health_check, + ): + repo = MagicMock() + repo.table.find_unique = AsyncMock(return_value=team_row) + MockTeamRepo.return_value = repo + with pytest.raises(HTTPException) as exc_info: + await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "openai/gpt-4o"}, + model_info={"id": "team-a-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert exc_info.value.status_code == 403 + key_check.assert_not_awaited() + health_check.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_test_model_connection_tolerates_missing_user_record_for_non_admin(): + """A key whose user record is gone is judged on the key's own model access.""" + from litellm.proxy.auth.auth_checks import UserNotFoundError + + mock_router = MagicMock() + mock_router.get_deployment.return_value = _configured_non_team_deployment() + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)), + patch( + "litellm.proxy.auth.auth_checks.get_user_object", + AsyncMock(side_effect=UserNotFoundError("gone")), + ), + patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)) as user_check, + patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})), + ): + result = await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "openai/gpt-4o"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert result["status"] == "success" + assert user_check.await_args.kwargs["user_object"] is None From 9b72cf720b39b9e5daf16e596dc0ce99f280631a Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 16:03:35 +0530 Subject: [PATCH 3/6] fix(proxy): bind the probe mode to the configured deployment for non-admins `mode` selects which provider operation the health probe performs, so it belongs to the configuration match like every other request value: a non-admin may only probe with the deployment's configured mode, or with none and let it be detected. Also split the authorization helper into a gate, the key/user access check and the orchestrator, and move the response filtering out of the handler, keeping each function's complexity within the repository's ceiling. --- .../health_endpoints/_health_endpoints.py | 143 ++++++++++++------ .../health_endpoints/test_health_endpoints.py | 42 ++++- 2 files changed, 135 insertions(+), 50 deletions(-) diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index ffb48a460b1..dafe2ad377e 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -2006,62 +2006,53 @@ async def health_liveliness_options(): _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS: Final[frozenset[str]] = frozenset(("model", "error", "mode_error")) -async def _authorize_test_connection( +def _test_connection_result_for_display(endpoint_data: dict, *, outcome_only: bool) -> dict: + """Clean the probe result for display; ``outcome_only`` hides the deployment's configuration.""" + cleaned: Final = _clean_endpoint_data(endpoint_data, details=True) + if not outcome_only: + return cleaned + return {k: v for k, v in cleaned.items() if k in _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS} + + +def _probe_is_configured_deployment( *, model_params: "Deployment", + configured_litellm_params: Mapping[str, object], + configured_mode: str | None, + request_litellm_params: Mapping[str, object], + requested_mode: str | None, +) -> bool: + """Whether a probe targets a team-less configured deployment exactly as configured. + + A request value that differs from the configuration (model, provider, + endpoint, credentials, mode, ...) describes a different probe. + """ + if getattr(model_params.model_info, "team_id", None) is not None: + return False + if any(configured_litellm_params.get(key) != value for key, value in request_litellm_params.items()): + return False + return requested_mode is None or requested_mode == configured_mode + + +async def _assert_caller_can_call_model( + *, + model: str, user_api_key_dict: UserAPIKeyAuth, prisma_client: "PrismaClient", - premium_user: bool, llm_router: "Router | None", - configured_model_name: str | None, - configured_litellm_params: Mapping[str, object], - request_litellm_params: Mapping[str, object], -) -> bool: - """Decide whether the caller may probe this model. - - Proxy admins and team admins may probe any model they manage, as before. - Any other user may probe a configured deployment that has no team, exactly - as configured, when their key and user are allowed to call its model: - a request value that differs from the configuration (model, provider, - endpoint, credentials, ...) describes a different probe, and that stays a - management operation. Team deployments keep their team-admin policy. - - Returns True when the caller was admitted through that non-admin path, so - the response can be limited to the outcome of the probe. - """ +) -> None: + """Raise 403 unless the caller's key and user may call ``model``.""" from litellm.proxy.auth.auth_checks import ( UserNotFoundError, can_key_call_model, can_user_call_model, get_user_object, ) - from litellm.proxy.management_endpoints.model_management_endpoints import ( - ModelManagementAuthChecks, - ) from litellm.proxy.proxy_server import llm_model_list, user_api_key_cache - try: - await ModelManagementAuthChecks.can_user_make_model_call( - model_params=model_params, - user_api_key_dict=user_api_key_dict, - prisma_client=prisma_client, - premium_user=premium_user, - ) - return False - except HTTPException as management_denial: - if management_denial.status_code != 403: - raise - if configured_model_name is None or getattr(model_params.model_info, "team_id", None) is not None: - raise - if any( - key != "mode" and configured_litellm_params.get(key) != value - for key, value in request_litellm_params.items() - ): - raise - try: await can_key_call_model( - model=configured_model_name, + model=model, llm_model_list=llm_model_list, valid_token=user_api_key_dict, llm_router=llm_router, @@ -2075,13 +2066,65 @@ async def _authorize_test_connection( ) except UserNotFoundError: user_object = None - await can_user_call_model( - model=configured_model_name, - llm_router=llm_router, - user_object=user_object, - ) + await can_user_call_model(model=model, llm_router=llm_router, user_object=user_object) except ProxyException as e: raise HTTPException(status_code=403, detail={"error": str(e.message)}) from e + + +async def _authorize_test_connection( + *, + model_params: "Deployment", + user_api_key_dict: UserAPIKeyAuth, + prisma_client: "PrismaClient", + premium_user: bool, + llm_router: "Router | None", + configured_model_name: str | None, + configured_litellm_params: Mapping[str, object], + configured_mode: str | None, + request_litellm_params: Mapping[str, object], + requested_mode: str | None, +) -> bool: + """Decide whether the caller may probe this model. + + Proxy admins and team admins may probe any model they manage, as before. + Any other user may probe a configured deployment that has no team, exactly + as configured, when their key and user are allowed to call its model. + Anything else stays a management operation. + + Returns True when the caller was admitted through that non-admin path, so + the response can be limited to the outcome of the probe. + """ + from litellm.proxy.management_endpoints.model_management_endpoints import ( + ModelManagementAuthChecks, + ) + + try: + await ModelManagementAuthChecks.can_user_make_model_call( + model_params=model_params, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, + premium_user=premium_user, + ) + return False + except HTTPException as management_denial: + if ( + management_denial.status_code != 403 + or configured_model_name is None + or not _probe_is_configured_deployment( + model_params=model_params, + configured_litellm_params=configured_litellm_params, + configured_mode=configured_mode, + request_litellm_params=request_litellm_params, + requested_mode=requested_mode, + ) + ): + raise + await _assert_caller_can_call_model( + model=configured_model_name, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, + llm_router=llm_router, + ) return True @@ -2278,7 +2321,9 @@ async def test_model_connection( llm_router=llm_router, configured_model_name=configured_model_name, configured_litellm_params=config_litellm_params, + configured_mode=(loaded_model_info or {}).get("mode"), request_litellm_params=request_litellm_params, + requested_mode=mode or request_litellm_params.get("mode"), ) mode = mode or litellm_params.pop("mode", None) @@ -2293,9 +2338,9 @@ async def test_model_connection( ) # Clean the result for display - cleaned_result = _clean_endpoint_data({**litellm_params, **result}, details=True) - if admitted_as_caller: - cleaned_result = {k: v for k, v in cleaned_result.items() if k in _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS} + cleaned_result: Final = _test_connection_result_for_display( + {**litellm_params, **result}, outcome_only=admitted_as_caller + ) return { "status": "error" if "error" in result else "success", diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index 543c4e2453d..c6a3aa6d5f0 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -4191,7 +4191,7 @@ def _configured_non_team_deployment(): api_base="https://configured.invalid/v1", api_key="CONFIGURED-API-KEY", ), - model_info=ModelInfo(id="non-team-deployment-id"), + model_info=ModelInfo(id="non-team-deployment-id", mode="chat"), ) @@ -4421,3 +4421,43 @@ async def test_test_model_connection_tolerates_missing_user_record_for_non_admin assert result["status"] == "success" assert user_check.await_args.kwargs["user_object"] is None + + +@pytest.mark.asyncio +async def test_test_model_connection_keeps_mode_overrides_admin_only(): + """ + `mode` selects which provider operation the probe performs, so a non-admin + may only probe with the deployment's configured mode (or none, which + auto-detects it). + """ + from fastapi import HTTPException + + from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo + + deployment = Deployment( + model_name="gpt-4o", + litellm_params=LiteLLM_Params(model="openai/gpt-4o", api_key="CONFIGURED-API-KEY"), + model_info=ModelInfo(id="non-team-deployment-id", mode="chat"), + ) + mock_router = MagicMock() + mock_router.get_deployment.return_value = deployment + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, + patch("litellm.ahealth_check", AsyncMock()) as health_check, + pytest.raises(HTTPException) as exc_info, + ): + await health_test_model_connection( + request=MagicMock(), + mode="image_generation", + litellm_params={"model": "openai/gpt-4o"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert exc_info.value.status_code == 403 + key_check.assert_not_awaited() + health_check.assert_not_awaited() From f7a5fb8467629bb85b11385ff843312548d526e1 Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 16:21:09 +0530 Subject: [PATCH 4/6] fix(proxy): accept an inferred probe mode and drop the added commentary A deployment without model_info.mode gets its probe mode from litellm.model_cost inside ahealth_check, so compare a non-admin's explicit mode against that same lookup instead of rejecting it. Use Mapping types and mark the two unavoidable mutable constructions, and remove the explanatory comments and docstrings per the repository's comment policy. --- .../health_endpoints/_health_endpoints.py | 47 +++++++-------- .../health_endpoints/test_health_endpoints.py | 59 +++++++++++-------- 2 files changed, 55 insertions(+), 51 deletions(-) diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index dafe2ad377e..2b08daf03cf 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -2001,17 +2001,26 @@ async def health_liveliness_options(): return Response(headers=response_headers, status_code=200) -# What a caller admitted through the non-admin path gets to see of the probe: -# the model and the outcome, none of the deployment's routing configuration. _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS: Final[frozenset[str]] = frozenset(("model", "error", "mode_error")) -def _test_connection_result_for_display(endpoint_data: dict, *, outcome_only: bool) -> dict: - """Clean the probe result for display; ``outcome_only`` hides the deployment's configuration.""" - cleaned: Final = _clean_endpoint_data(endpoint_data, details=True) +def _test_connection_result_for_display( + litellm_params: Mapping[str, object], result: Mapping[str, object], *, outcome_only: bool +) -> Mapping[str, object]: + cleaned: Final = _clean_endpoint_data({**litellm_params, **result}, details=True) if not outcome_only: return cleaned - return {k: v for k, v in cleaned.items() if k in _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS} + return { # mutable-ok: fresh filtered copy handed to the caller + k: v for k, v in cleaned.items() if k in _NON_ADMIN_TEST_CONNECTION_RESULT_KEYS + } + + +def _configured_probe_mode(model_info: Mapping[str, object] | None, model: object) -> str | None: + configured: Final = model_info.get("mode") if model_info else None + if configured is not None: + return str(configured) + cost_entry: Final = litellm.model_cost.get(model) if isinstance(model, str) else None + return cost_entry.get("mode") if cost_entry else None def _probe_is_configured_deployment( @@ -2022,11 +2031,6 @@ def _probe_is_configured_deployment( request_litellm_params: Mapping[str, object], requested_mode: str | None, ) -> bool: - """Whether a probe targets a team-less configured deployment exactly as configured. - - A request value that differs from the configuration (model, provider, - endpoint, credentials, mode, ...) describes a different probe. - """ if getattr(model_params.model_info, "team_id", None) is not None: return False if any(configured_litellm_params.get(key) != value for key, value in request_litellm_params.items()): @@ -2041,7 +2045,6 @@ async def _assert_caller_can_call_model( prisma_client: "PrismaClient", llm_router: "Router | None", ) -> None: - """Raise 403 unless the caller's key and user may call ``model``.""" from litellm.proxy.auth.auth_checks import ( UserNotFoundError, can_key_call_model, @@ -2068,7 +2071,10 @@ async def _assert_caller_can_call_model( user_object = None await can_user_call_model(model=model, llm_router=llm_router, user_object=user_object) except ProxyException as e: - raise HTTPException(status_code=403, detail={"error": str(e.message)}) from e + raise HTTPException( + status_code=403, + detail={"error": str(e.message)}, # mutable-ok: same 403 payload shape as the rest of this endpoint + ) from e async def _authorize_test_connection( @@ -2084,16 +2090,7 @@ async def _authorize_test_connection( request_litellm_params: Mapping[str, object], requested_mode: str | None, ) -> bool: - """Decide whether the caller may probe this model. - - Proxy admins and team admins may probe any model they manage, as before. - Any other user may probe a configured deployment that has no team, exactly - as configured, when their key and user are allowed to call its model. - Anything else stays a management operation. - - Returns True when the caller was admitted through that non-admin path, so - the response can be limited to the outcome of the probe. - """ + """Returns True when a non-admin was admitted to probe a team-less deployment as configured.""" from litellm.proxy.management_endpoints.model_management_endpoints import ( ModelManagementAuthChecks, ) @@ -2321,7 +2318,7 @@ async def test_model_connection( llm_router=llm_router, configured_model_name=configured_model_name, configured_litellm_params=config_litellm_params, - configured_mode=(loaded_model_info or {}).get("mode"), + configured_mode=_configured_probe_mode(loaded_model_info, config_litellm_params.get("model")), request_litellm_params=request_litellm_params, requested_mode=mode or request_litellm_params.get("mode"), ) @@ -2339,7 +2336,7 @@ async def test_model_connection( # Clean the result for display cleaned_result: Final = _test_connection_result_for_display( - {**litellm_params, **result}, outcome_only=admitted_as_caller + litellm_params, result, outcome_only=admitted_as_caller ) return { diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index c6a3aa6d5f0..b9b05c37f10 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -4205,10 +4205,6 @@ def _internal_user(): @pytest.mark.asyncio async def test_test_model_connection_allows_internal_user_to_probe_configured_model_they_can_call(): - """ - An internal user may test a configured, non-team model they are allowed - to call, and the probe runs with the configured credentials. - """ mock_router = MagicMock() mock_router.get_deployment.return_value = _configured_non_team_deployment() @@ -4233,7 +4229,6 @@ async def test_test_model_connection_allows_internal_user_to_probe_configured_mo assert key_check.await_args.kwargs["model"] == "gpt-4o" assert user_check.await_args.kwargs["model"] == "gpt-4o" assert health_check.await_args.kwargs["model_params"]["api_key"] == "CONFIGURED-API-KEY" - # the caller learns the outcome, not the deployment's routing configuration assert "api_base" not in result["result"] assert "api_key" not in result["result"] @@ -4278,11 +4273,6 @@ async def test_test_model_connection_denies_internal_user_without_model_access() async def test_test_model_connection_keeps_connection_overrides_admin_only(): from fastapi import HTTPException - """ - A request that sets its own connection fields describes a different - endpoint than the configured one; probing that stays a management - operation, so an internal user is denied before any access check runs. - """ mock_router = MagicMock() mock_router.get_deployment.return_value = _configured_non_team_deployment() @@ -4312,11 +4302,6 @@ async def test_test_model_connection_keeps_connection_overrides_admin_only(): @pytest.mark.asyncio async def test_test_model_connection_keeps_model_overrides_admin_only(): - """ - The probe runs whatever `litellm_params.model` the request sends, so a - non-admin may only send the configured deployment's own model: the access - check and the probe must target the same model. - """ from fastapi import HTTPException mock_router = MagicMock() @@ -4345,11 +4330,6 @@ async def test_test_model_connection_keeps_model_overrides_admin_only(): @pytest.mark.asyncio async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_admins(): - """ - A team deployment keeps its team-admin policy: a non-admin whose key can - call the same model name must not reach the non-admin path with the - team's configured credentials. - """ from fastapi import HTTPException from litellm.proxy._types import LiteLLM_TeamTable @@ -4393,7 +4373,6 @@ async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_a @pytest.mark.asyncio async def test_test_model_connection_tolerates_missing_user_record_for_non_admin(): - """A key whose user record is gone is judged on the key's own model access.""" from litellm.proxy.auth.auth_checks import UserNotFoundError mock_router = MagicMock() @@ -4425,11 +4404,6 @@ async def test_test_model_connection_tolerates_missing_user_record_for_non_admin @pytest.mark.asyncio async def test_test_model_connection_keeps_mode_overrides_admin_only(): - """ - `mode` selects which provider operation the probe performs, so a non-admin - may only probe with the deployment's configured mode (or none, which - auto-detects it). - """ from fastapi import HTTPException from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo @@ -4461,3 +4435,36 @@ async def test_test_model_connection_keeps_mode_overrides_admin_only(): assert exc_info.value.status_code == 403 key_check.assert_not_awaited() health_check.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_test_model_connection_accepts_mode_the_probe_would_infer(): + from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo + + deployment = Deployment( + model_name="gpt-4o", + litellm_params=LiteLLM_Params(model="gpt-4o", api_key="CONFIGURED-API-KEY"), + model_info=ModelInfo(id="non-team-deployment-id"), + ) + mock_router = MagicMock() + mock_router.get_deployment.return_value = deployment + + with ( + patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)), + patch("litellm.proxy.auth.auth_checks.get_user_object", AsyncMock(return_value=None)), + patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)), + patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})) as health_check, + ): + result = await health_test_model_connection( + request=MagicMock(), + mode="chat", + litellm_params={"model": "gpt-4o"}, + model_info={"id": "non-team-deployment-id"}, + user_api_key_dict=_internal_user(), + ) + + assert result["status"] == "success" + assert health_check.await_args.kwargs["mode"] == "chat" From 36912cf03f8526c946ed404ecc18ce21bfcafb70 Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 16:49:35 +0530 Subject: [PATCH 5/6] test(proxy): share the probe environment across the test_connection tests The eight internal-user tests each patched the same proxy_server globals and access checks. Move that into one context manager, marked with the reasons the test-quality gate asks for, and fold the three override denials into a parametrized test. --- .../health_endpoints/test_health_endpoints.py | 286 ++++++++---------- 1 file changed, 119 insertions(+), 167 deletions(-) diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index b9b05c37f10..808c863d560 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -3,7 +3,7 @@ import copy import json import time from collections.abc import Iterator, Mapping, Sequence -from contextlib import contextmanager +from contextlib import ExitStack, contextmanager from datetime import datetime, timedelta from types import SimpleNamespace from typing import Final @@ -4181,7 +4181,7 @@ async def test_health_services_endpoint_pointfive_blocks_non_admin(monkeypatch, logger_class.assert_not_called() -def _configured_non_team_deployment(): +def _configured_non_team_deployment(mode: str | None = "chat"): from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo return Deployment( @@ -4191,7 +4191,7 @@ def _configured_non_team_deployment(): api_base="https://configured.invalid/v1", api_key="CONFIGURED-API-KEY", ), - model_info=ModelInfo(id="non-team-deployment-id", mode="chat"), + model_info=ModelInfo(id="non-team-deployment-id", mode=mode), ) @@ -4203,27 +4203,84 @@ def _internal_user(): ) +@contextmanager +def _probe_environment( + deployment, + *, + key_check: AsyncMock | None = None, + user_lookup: AsyncMock | None = None, + user_check: AsyncMock | None = None, + health_check: AsyncMock | None = None, +) -> Iterator[None]: + mock_router = MagicMock() + mock_router.get_deployment.return_value = deployment + with ExitStack() as stack: + stack.enter_context( + patch( # test-quality-ok: the endpoint reads proxy_server globals, stubbed like the rest of this file + "litellm.proxy.proxy_server.prisma_client", MagicMock() + ) + ) + stack.enter_context( + patch( # test-quality-ok: the endpoint reads proxy_server globals, stubbed like the rest of this file + "litellm.proxy.proxy_server.llm_router", mock_router + ) + ) + stack.enter_context( + patch( # test-quality-ok: the endpoint reads proxy_server globals, stubbed like the rest of this file + "litellm.proxy.proxy_server.premium_user", True + ) + ) + if key_check is not None: + stack.enter_context( + patch( # test-quality-ok: the access check needs a live router and DB, stubbed like the rest of this file + "litellm.proxy.auth.auth_checks.can_key_call_model", key_check + ) + ) + if user_lookup is not None: + stack.enter_context( + patch( # test-quality-ok: the user lookup needs a live DB, stubbed like the rest of this file + "litellm.proxy.auth.auth_checks.get_user_object", user_lookup + ) + ) + if user_check is not None: + stack.enter_context( + patch( # test-quality-ok: the access check needs a live router and DB, stubbed like the rest of this file + "litellm.proxy.auth.auth_checks.can_user_call_model", user_check + ) + ) + if health_check is not None: + stack.enter_context( + patch( # test-quality-ok: the probe is a real provider call, stubbed like the rest of this file + "litellm.ahealth_check", health_check + ) + ) + yield + + +async def _probe_as_internal_user(litellm_params, mode="chat", deployment_id="non-team-deployment-id"): + return await health_test_model_connection( + request=MagicMock(), + mode=mode, + litellm_params=litellm_params, + model_info={"id": deployment_id}, + user_api_key_dict=_internal_user(), + ) + + @pytest.mark.asyncio async def test_test_model_connection_allows_internal_user_to_probe_configured_model_they_can_call(): - mock_router = MagicMock() - mock_router.get_deployment.return_value = _configured_non_team_deployment() + key_check = AsyncMock(return_value=True) + user_check = AsyncMock(return_value=True) + health_check = AsyncMock(return_value={"status": "healthy"}) - with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, - patch("litellm.proxy.auth.auth_checks.get_user_object", AsyncMock(return_value=None)), - patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)) as user_check, - patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})) as health_check, + with _probe_environment( + _configured_non_team_deployment(), + key_check=key_check, + user_lookup=AsyncMock(return_value=None), + user_check=user_check, + health_check=health_check, ): - result = await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "openai/gpt-4o"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + result = await _probe_as_internal_user({"model": "openai/gpt-4o"}) assert result["status"] == "success" assert key_check.await_args.kwargs["model"] == "gpt-4o" @@ -4239,30 +4296,21 @@ async def test_test_model_connection_denies_internal_user_without_model_access() from litellm.proxy._types import ProxyErrorTypes, ProxyException - mock_router = MagicMock() - mock_router.get_deployment.return_value = _configured_non_team_deployment() denied = ProxyException( message="Key not allowed to access model. Tried to access gpt-4o", type=ProxyErrorTypes.key_model_access_denied, param="model", code=403, ) + health_check = AsyncMock() with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(side_effect=denied)), - patch("litellm.ahealth_check", AsyncMock()) as health_check, + _probe_environment( + _configured_non_team_deployment(), key_check=AsyncMock(side_effect=denied), health_check=health_check + ), pytest.raises(HTTPException) as exc_info, ): - await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "openai/gpt-4o"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + await _probe_as_internal_user({"model": "openai/gpt-4o"}) assert exc_info.value.status_code == 403 assert "not allowed to access model" in exc_info.value.detail["error"] @@ -4270,58 +4318,26 @@ async def test_test_model_connection_denies_internal_user_without_model_access() @pytest.mark.asyncio -async def test_test_model_connection_keeps_connection_overrides_admin_only(): +@pytest.mark.parametrize( + "litellm_params, mode", + [ + ({"model": "openai/gpt-4o", "api_base": "https://somewhere-else.invalid/v1"}, "chat"), + ({"model": "openai/gpt-4o-mini"}, "chat"), + ({"model": "openai/gpt-4o"}, "image_generation"), + ], + ids=["api_base", "model", "mode"], +) +async def test_test_model_connection_keeps_overrides_admin_only(litellm_params, mode): from fastapi import HTTPException - mock_router = MagicMock() - mock_router.get_deployment.return_value = _configured_non_team_deployment() + key_check = AsyncMock(return_value=True) + health_check = AsyncMock() with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, - patch("litellm.ahealth_check", AsyncMock()) as health_check, + _probe_environment(_configured_non_team_deployment(), key_check=key_check, health_check=health_check), pytest.raises(HTTPException) as exc_info, ): - await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={ - "model": "openai/gpt-4o", - "api_base": "https://somewhere-else.invalid/v1", - }, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) - - assert exc_info.value.status_code == 403 - key_check.assert_not_awaited() - health_check.assert_not_awaited() - - -@pytest.mark.asyncio -async def test_test_model_connection_keeps_model_overrides_admin_only(): - from fastapi import HTTPException - - mock_router = MagicMock() - mock_router.get_deployment.return_value = _configured_non_team_deployment() - - with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, - patch("litellm.ahealth_check", AsyncMock()) as health_check, - pytest.raises(HTTPException) as exc_info, - ): - await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "openai/gpt-4o-mini"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + await _probe_as_internal_user(litellm_params, mode=mode) assert exc_info.value.status_code == 403 key_check.assert_not_awaited() @@ -4340,31 +4356,23 @@ async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_a litellm_params=LiteLLM_Params(model="openai/gpt-4o", api_key="TEAM-A-API-KEY"), model_info=ModelInfo(id="team-a-deployment-id", team_id="team-a"), ) - mock_router = MagicMock() - mock_router.get_deployment.return_value = team_deployment team_row = SimpleNamespace( model_dump=lambda: LiteLLM_TeamTable(team_id="team-a", members_with_roles=[]).model_dump() ) + key_check = AsyncMock(return_value=True) + health_check = AsyncMock() with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.management_endpoints.model_management_endpoints.TeamRepository") as MockTeamRepo, - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, - patch("litellm.ahealth_check", AsyncMock()) as health_check, + _probe_environment(team_deployment, key_check=key_check, health_check=health_check), + patch( # test-quality-ok: the team lookup needs a live DB, stubbed like the surrounding team tests + "litellm.proxy.management_endpoints.model_management_endpoints.TeamRepository" + ) as MockTeamRepo, ): repo = MagicMock() repo.table.find_unique = AsyncMock(return_value=team_row) MockTeamRepo.return_value = repo with pytest.raises(HTTPException) as exc_info: - await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "openai/gpt-4o"}, - model_info={"id": "team-a-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + await _probe_as_internal_user({"model": "openai/gpt-4o"}, deployment_id="team-a-deployment-id") assert exc_info.value.status_code == 403 key_check.assert_not_awaited() @@ -4375,68 +4383,21 @@ async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_a async def test_test_model_connection_tolerates_missing_user_record_for_non_admin(): from litellm.proxy.auth.auth_checks import UserNotFoundError - mock_router = MagicMock() - mock_router.get_deployment.return_value = _configured_non_team_deployment() + user_check = AsyncMock(return_value=True) - with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)), - patch( - "litellm.proxy.auth.auth_checks.get_user_object", - AsyncMock(side_effect=UserNotFoundError("gone")), - ), - patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)) as user_check, - patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})), + with _probe_environment( + _configured_non_team_deployment(), + key_check=AsyncMock(return_value=True), + user_lookup=AsyncMock(side_effect=UserNotFoundError("gone")), + user_check=user_check, + health_check=AsyncMock(return_value={"status": "healthy"}), ): - result = await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "openai/gpt-4o"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + result = await _probe_as_internal_user({"model": "openai/gpt-4o"}) assert result["status"] == "success" assert user_check.await_args.kwargs["user_object"] is None -@pytest.mark.asyncio -async def test_test_model_connection_keeps_mode_overrides_admin_only(): - from fastapi import HTTPException - - from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo - - deployment = Deployment( - model_name="gpt-4o", - litellm_params=LiteLLM_Params(model="openai/gpt-4o", api_key="CONFIGURED-API-KEY"), - model_info=ModelInfo(id="non-team-deployment-id", mode="chat"), - ) - mock_router = MagicMock() - mock_router.get_deployment.return_value = deployment - - with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)) as key_check, - patch("litellm.ahealth_check", AsyncMock()) as health_check, - pytest.raises(HTTPException) as exc_info, - ): - await health_test_model_connection( - request=MagicMock(), - mode="image_generation", - litellm_params={"model": "openai/gpt-4o"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) - - assert exc_info.value.status_code == 403 - key_check.assert_not_awaited() - health_check.assert_not_awaited() - - @pytest.mark.asyncio async def test_test_model_connection_accepts_mode_the_probe_would_infer(): from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo @@ -4446,25 +4407,16 @@ async def test_test_model_connection_accepts_mode_the_probe_would_infer(): litellm_params=LiteLLM_Params(model="gpt-4o", api_key="CONFIGURED-API-KEY"), model_info=ModelInfo(id="non-team-deployment-id"), ) - mock_router = MagicMock() - mock_router.get_deployment.return_value = deployment + health_check = AsyncMock(return_value={"status": "healthy"}) - with ( - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), - patch("litellm.proxy.proxy_server.llm_router", mock_router), - patch("litellm.proxy.proxy_server.premium_user", True), - patch("litellm.proxy.auth.auth_checks.can_key_call_model", AsyncMock(return_value=True)), - patch("litellm.proxy.auth.auth_checks.get_user_object", AsyncMock(return_value=None)), - patch("litellm.proxy.auth.auth_checks.can_user_call_model", AsyncMock(return_value=True)), - patch("litellm.ahealth_check", AsyncMock(return_value={"status": "healthy"})) as health_check, + with _probe_environment( + deployment, + key_check=AsyncMock(return_value=True), + user_lookup=AsyncMock(return_value=None), + user_check=AsyncMock(return_value=True), + health_check=health_check, ): - result = await health_test_model_connection( - request=MagicMock(), - mode="chat", - litellm_params={"model": "gpt-4o"}, - model_info={"id": "non-team-deployment-id"}, - user_api_key_dict=_internal_user(), - ) + result = await _probe_as_internal_user({"model": "gpt-4o"}) assert result["status"] == "success" assert health_check.await_args.kwargs["mode"] == "chat" From 03ee54536613ec32aec571bf3e3e6dec4281e0b0 Mon Sep 17 00:00:00 2001 From: Ishkirat-Singh Date: Wed, 9 Sep 2026 18:25:44 +0530 Subject: [PATCH 6/6] test(proxy): type the shared test_connection helpers --- .../health_endpoints/test_health_endpoints.py | 19 +++++++++---------- 1 file changed, 9 insertions(+), 10 deletions(-) diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index 808c863d560..fd37cfcff2f 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -33,6 +33,7 @@ from litellm.proxy.health_endpoints._health_endpoints import ( from litellm.proxy.health_endpoints._health_endpoints import ( test_model_connection as health_test_model_connection, ) +from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo # Import shared proxy test helpers from conftest from tests.test_litellm.proxy.conftest import create_proxy_test_client @@ -4181,9 +4182,7 @@ async def test_health_services_endpoint_pointfive_blocks_non_admin(monkeypatch, logger_class.assert_not_called() -def _configured_non_team_deployment(mode: str | None = "chat"): - from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo - +def _configured_non_team_deployment(mode: str | None = "chat") -> Deployment: return Deployment( model_name="gpt-4o", litellm_params=LiteLLM_Params( @@ -4195,7 +4194,7 @@ def _configured_non_team_deployment(mode: str | None = "chat"): ) -def _internal_user(): +def _internal_user() -> UserAPIKeyAuth: return UserAPIKeyAuth( token="internal-user-token", user_id="internal-user", @@ -4205,7 +4204,7 @@ def _internal_user(): @contextmanager def _probe_environment( - deployment, + deployment: Deployment, *, key_check: AsyncMock | None = None, user_lookup: AsyncMock | None = None, @@ -4257,7 +4256,9 @@ def _probe_environment( yield -async def _probe_as_internal_user(litellm_params, mode="chat", deployment_id="non-team-deployment-id"): +async def _probe_as_internal_user( + litellm_params: dict[str, object], mode: str | None = "chat", deployment_id: str = "non-team-deployment-id" +) -> dict[str, object]: return await health_test_model_connection( request=MagicMock(), mode=mode, @@ -4327,7 +4328,7 @@ async def test_test_model_connection_denies_internal_user_without_model_access() ], ids=["api_base", "model", "mode"], ) -async def test_test_model_connection_keeps_overrides_admin_only(litellm_params, mode): +async def test_test_model_connection_keeps_overrides_admin_only(litellm_params: dict[str, object], mode: str) -> None: from fastapi import HTTPException key_check = AsyncMock(return_value=True) @@ -4349,7 +4350,6 @@ async def test_test_model_connection_keeps_team_deployments_admin_only_for_non_a from fastapi import HTTPException from litellm.proxy._types import LiteLLM_TeamTable - from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo team_deployment = Deployment( model_name="gpt-4o", @@ -4399,8 +4399,7 @@ async def test_test_model_connection_tolerates_missing_user_record_for_non_admin @pytest.mark.asyncio -async def test_test_model_connection_accepts_mode_the_probe_would_infer(): - from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo +async def test_test_model_connection_accepts_mode_the_probe_would_infer() -> None: deployment = Deployment( model_name="gpt-4o",