diff --git a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py index aa11ae59328..6613690f448 100644 --- a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py +++ b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py @@ -268,6 +268,10 @@ class MCPRequestHandler: ) if ( mcp_servers_from_path is not None + and not _has_client_supplied_mcp_auth( + mcp_auth_header, + mcp_server_auth_headers, + ) and _is_mcp_passthrough_cold_start( mcp_servers_from_path, client_ip=client_ip ) diff --git a/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py b/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py index c535b9b98fa..0529c45e6fe 100644 --- a/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py @@ -31,8 +31,11 @@ from litellm.types.mcp_server.mcp_server_manager import MCPServer # TTL cache for upstream OAuth metadata fetched from pass-through MCP servers. # Keeps us from hammering the upstream IdP on each discovery request. # Keyed by (server_id, resource_url) → (expires_at_epoch, payload). -_OAUTH_METADATA_CACHE: Dict[Tuple[str, str], Tuple[float, dict]] = {} +# A payload of ``None`` is a negative-result entry that prevents repeated +# upstream fetches when the IdP consistently has no metadata to serve. +_OAUTH_METADATA_CACHE: Dict[Tuple[str, str], Tuple[float, Optional[dict]]] = {} _OAUTH_METADATA_CACHE_TTL_SECONDS = 300 +_OAUTH_METADATA_NEGATIVE_CACHE_TTL_SECONDS = 60 _OAUTH_METADATA_CACHE_MAX_SIZE = 128 # Per-(server_id, resource_url) async locks so concurrent discovery requests # coalesce onto a single upstream fetch instead of issuing N parallel calls. @@ -830,6 +833,15 @@ async def fetch_upstream_oauth_protected_resource( if len(network_errors) == len(candidates): raise network_errors[-1] + # Negative-result caching: when no candidate yielded a usable payload, + # remember that for a shorter TTL so we don't re-fetch on every + # subsequent discovery request (and so the per-key lock can be pruned). + now = time.time() + _OAUTH_METADATA_CACHE[cache_key] = ( + now + _OAUTH_METADATA_NEGATIVE_CACHE_TTL_SECONDS, + None, + ) + _prune_oauth_metadata_cache(now) return None diff --git a/litellm/proxy/auth/auth_utils.py b/litellm/proxy/auth/auth_utils.py index 4f83fe570b5..5c24aca632c 100644 --- a/litellm/proxy/auth/auth_utils.py +++ b/litellm/proxy/auth/auth_utils.py @@ -518,7 +518,8 @@ def get_request_route(request: Request) -> str: if root_path and ( raw_path == root_path or raw_path.startswith(root_path + "/") ): - return raw_path[len(root_path) :] + stripped = raw_path[len(root_path) :] + return stripped or "/" return raw_path except Exception as e: verbose_proxy_logger.debug(