diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py index 1bd23db0948..6a5e55d1822 100644 --- a/litellm/proxy/_experimental/mcp_server/server.py +++ b/litellm/proxy/_experimental/mcp_server/server.py @@ -213,7 +213,7 @@ def _request_tags_from_raw_headers( ) -> Sequence[str] | None: """The caller's tags, parsed by the same helper the LLM routes use so an MCP operation and a chat completion attribute an identical header identically.""" - header_value = _request_tags_header(raw_headers) + header_value: Final = _request_tags_header(raw_headers) if header_value is None: return None return LiteLLMProxyRequestSetup.add_request_tag_to_metadata( @@ -2190,7 +2190,11 @@ if MCP_AVAILABLE: list_tools_call_id: Final = str(uuid.uuid4()) # Derive trace_id from raw_headers when not explicitly passed (same as A2A / MCP call_tool) effective_litellm_trace_id: Final = litellm_trace_id or get_chain_id_from_headers(raw_headers) - effective_request_tags: Final = request_tags or _request_tags_from_raw_headers(raw_headers) + # An explicit [] means the caller resolved to no tags; only fall back to the + # header when nothing was passed at all. + effective_request_tags: Final = ( + request_tags if request_tags is not None else _request_tags_from_raw_headers(raw_headers) + ) spend_logs_metadata: Final[dict[str, object]] = { "mcp_operation": "list_tools", } diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py index 09d7592d4d4..1b68303cfdb 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py @@ -5031,26 +5031,26 @@ async def test_get_tools_from_mcp_servers_takes_list_tools_tags_from_x_litellm_t return dummy_logging_obj, None with ( - patch( + patch( # test-quality-ok: server allowlist is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers", new=AsyncMock(return_value=[server_a]), ), - patch( + patch( # test-quality-ok: header prep is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers", return_value=(None, None), ), - patch( + patch( # test-quality-ok: manager is a module-level singleton; patching it is the suite established seam "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager", ) as mock_manager, - patch( + patch( # test-quality-ok: tool filter is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools", side_effect=lambda tools, _server: tools, ), - patch( + patch( # test-quality-ok: permission filter is a module-level async function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions", new=AsyncMock(side_effect=lambda tools, **_: tools), ), - patch( + patch( # test-quality-ok: logging setup is a module-level function; patched to capture spend-log metadata kwargs "litellm.proxy._experimental.mcp_server.server.function_setup", side_effect=_capture_function_setup, ), @@ -5073,7 +5073,8 @@ async def test_get_tools_from_mcp_servers_takes_list_tools_tags_from_x_litellm_t @pytest.mark.asyncio async def test_get_tools_from_mcp_servers_prefers_explicit_request_tags_over_the_header(): - """`request_tags` is the resolved value a caller passes in; a header must not override it.""" + """`request_tags` is the resolved value a caller passes in; a header must not override it. + An explicit empty list resolves to no tags rather than falling back to the header.""" try: from litellm.proxy._experimental.mcp_server.server import ( _get_tools_from_mcp_servers, @@ -5105,26 +5106,26 @@ async def test_get_tools_from_mcp_servers_prefers_explicit_request_tags_over_the return dummy_logging_obj, None with ( - patch( + patch( # test-quality-ok: server allowlist is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers", new=AsyncMock(return_value=[server_a]), ), - patch( + patch( # test-quality-ok: header prep is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers", return_value=(None, None), ), - patch( + patch( # test-quality-ok: manager is a module-level singleton; patching it is the suite established seam "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager", ) as mock_manager, - patch( + patch( # test-quality-ok: tool filter is a module-level function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools", side_effect=lambda tools, _server: tools, ), - patch( + patch( # test-quality-ok: permission filter is a module-level async function; the suite has no injection seam "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions", new=AsyncMock(side_effect=lambda tools, **_: tools), ), - patch( + patch( # test-quality-ok: logging setup is a module-level function; patched to capture spend-log metadata kwargs "litellm.proxy._experimental.mcp_server.server.function_setup", side_effect=_capture_function_setup, ), @@ -5141,8 +5142,22 @@ async def test_get_tools_from_mcp_servers_prefers_explicit_request_tags_over_the list_tools_log_source="mcp_protocol", request_tags=["explicit"], ) + explicit_metadata = dict(function_setup_kwargs["metadata"]) - assert function_setup_kwargs["metadata"]["tags"] == ["explicit"] + await _get_tools_from_mcp_servers( + user_api_key_auth=user_auth, + mcp_auth_header=None, + mcp_servers=["server_a"], + mcp_server_auth_headers=None, + raw_headers={"x-litellm-tags": "from-header"}, + log_list_tools_to_spendlogs=True, + list_tools_log_source="mcp_protocol", + request_tags=[], + ) + empty_metadata = dict(function_setup_kwargs["metadata"]) + + assert explicit_metadata["tags"] == ["explicit"] + assert "tags" not in empty_metadata @pytest.mark.parametrize( diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 7eadaa6c991..839aa52fa84 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -16781,7 +16781,6 @@ export interface paths { * - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking. * - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" } * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. - * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) @@ -16887,7 +16886,6 @@ export interface paths { * - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking. * - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" } * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. - * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)