diff --git a/tests/e2e/coverage_registry/README.md b/tests/e2e/coverage_registry/README.md index 4f9845bab87..30a6ebdf4bf 100644 --- a/tests/e2e/coverage_registry/README.md +++ b/tests/e2e/coverage_registry/README.md @@ -108,3 +108,38 @@ things to settle before treating the set as final: boundary needs a decision, and the auth cluster may deserve promotion to its own module - the P2 "niche" cells each stand in for a large tail of integrations/providers by design, so the denominator is deliberately P0-weighted rather than a full inventory + +## MCP migration acceptance (LIT-4506) + +This matrix tracks the ten security/compatibility guards and the added JWT acceptance +criteria. Registry membership means a desired behavior, not a passing execution +record. Preserve the source/harness commits, SDK version, auth and transport topology, +executed node IDs and live results with each PR. A skipped or unexecuted case remains +pending. Integration contract IDs belong in `tests/integration/contracts.json`, not +`mcp.yaml`; links below connect the two suites without merging their namespaces + +| Requirement | Existing regression or owning suite | Remaining acceptance and owner | Required before | +|---|---|---|---| +| Principal discovery | `mcp/test_mcp_key_access_e2e.py`, `mcp/test_mcp_access_group_e2e.py`, `mcp/test_mcp_toolset_enforcement_e2e.py` | Run key/team/org/user controls on every configured replica; the E2E health check intersects with test-owned servers and proves grant visibility only, non-disclosure of unrelated servers is proven by integration contract other.mcp.health.restricted_keys_intersect_grants_in_both_modes (#41731); native/REST parity remains LIT-4506 | Phase 0 and affected authorization changes | +| No self-attached unauthorized grants | Management key authorization tests | Read back unchanged server/toolset/access-group grants after rejected writes, LIT-4502 | Affected grant capability activation | +| UI/API parity | Admin MCP UI suite | Same non-admin actor and permissions across both surfaces, LIT-3644 / LIT-4506 | Affected UI capability activation | +| Server identity routing | Saved-server lifecycle integration and resolver tests | Cold routing, duplicate/unprefixed names, LIT-4500 | Routing changes | +| Same-URL credential isolation | Outbound credential resolver subject/server tests | Observe distinct user/server credentials at actual upstream transport, LIT-4506 with LIT-3467 / LIT-3559 | Affected auth capability activation | +| Fail-closed credentials | LIT-4501 production-path tests; integration lifecycle owns warm credential-removal follow-up | Observe rejection and zero upstream calls after persisted removal; refresh/challenge work remains LIT-4436 / LIT-4422 / LIT-3433 | Phase 0 auth release | +| Upstream session continuity | Existing controlled integration peer is stateless | Observe upstream session ID across operations, LIT-3143 | Stateful upstream activation | +| Real hooks/guardrails | `mcp/test_mcp_guardrail_e2e.py`, LIT-4889 production-path tests | Request-selected direct/virtual pre-call guard integration and removal of void assertions, LIT-4506 | Phase 0 and changed guardrail paths | +| Discovery and execution permissions | Key-denial and principal-toolset E2E; `integration/compatibility/test_persisted_toolsets.py` | Preserve permitted calls and explicit denial, expand remaining protocol combinations, LIT-4506 | Every authorization migration | +| Stateless/stateful combinations | Stateless SDK-peer integration | Explicit incoming/outgoing combinations and supported versions, LIT-3143 / LIT-3559 | Affected transport activation | +| JWT canonical owner and restart | LIT-3794 / LIT-3795 merged regressions | Real process restart, cold cache, header precedence, expired/invalid JWT and inactive user, LIT-4506 | Applicable Phase 0 auth release | +| No gateway JWT upstream | Existing credential resolver/unit evidence | Observe actual upstream headers across users and same-URL servers, LIT-4506 | Applicable Phase 0 auth release | +| Uninterrupted OAuth and aggregate SSO | `mcp/test_mcp_chat_completion_oauth_e2e.py` | Immediate SDK continuation and aggregate SSO, LIT-3467 with LIT-4506 acceptance; missing browser login is not a pass | Applicable Phase 0 auth release | + +The initial package does not close LIT-4506. Recheck all applicable rows before the +execution plan's final acceptance step. Keep protocol conformance, canary and rollback +evidence with their owners; a static registry percentage cannot authorize rollout + +The E2E consolidation incorporates PR #34055's inventory and PR #35405's remaining +Datadog input-schema guard. The telemetry argument removal and unskipping already +landed in #38640. LIT-5052 requires the schema guard and all three affected real-Datadog +tests, including seeded completion retrieval, to pass before closure. LIT-5749's runtime +fix already landed in #38488; the principal tests add regression evidence diff --git a/tests/e2e/coverage_registry/mcp.yaml b/tests/e2e/coverage_registry/mcp.yaml index 1cdeac7b77f..1671b4eba27 100644 --- a/tests/e2e/coverage_registry/mcp.yaml +++ b/tests/e2e/coverage_registry/mcp.yaml @@ -135,3 +135,352 @@ assertions: [toolset_scoped] source: "user_api_key_auth_mcp.py:2137" rationale: "A key granted a toolset lists exactly the toolset's tools: the rest of the server's catalog stays hidden and every stored name resolves" +- id: mcp.list_resource_templates.api_key.succeeds + module: mcp + tier: P2 + operation: list_resource_templates + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/_experimental/mcp_server/server.py:list_resource_templates + rationale: Resource-template listing; same auth stack as tools +- id: mcp.server_add.api_key.admin_only + module: mcp + tier: P0 + operation: server_add + auth_family: api_key + assertions: + - admin_only + - persists + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:add_mcp_server + rationale: PROXY_ADMIN-gated server registration; persists to DB and is picked up without restart; onboarding path +- id: mcp.server_list.api_key.succeeds + module: mcp + tier: P1 + operation: server_list + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:fetch_all_mcp_servers + rationale: List registered servers; admin required only to query another team +- id: mcp.server_get.api_key.succeeds + module: mcp + tier: P2 + operation: server_get + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:fetch_mcp_server + rationale: Fetch one server; non-admins filtered to their allowed servers +- id: mcp.server_update.api_key.admin_only + module: mcp + tier: P1 + operation: server_update + auth_family: api_key + assertions: + - admin_only + - persists + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:edit_mcp_server + rationale: PROXY_ADMIN-gated edit persists to DB +- id: mcp.server_delete.api_key.admin_only + module: mcp + tier: P1 + operation: server_delete + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:remove_mcp_server + rationale: PROXY_ADMIN-gated delete removes the server row +- id: mcp.server_health.api_key.reports_status + module: mcp + tier: P1 + operation: server_health + auth_family: api_key + assertions: + - reports_status + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:health_check_servers + rationale: Health check distinguishes a reachable upstream from a broken one +- id: mcp.server_submit.api_key.creates_pending + module: mcp + tier: P1 + operation: server_submit + auth_family: api_key + assertions: + - creates_pending + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:register_mcp_server + rationale: Non-admin submission creates a pending row for review; admins auto-approve +- id: mcp.server_approve.api_key.admin_only + module: mcp + tier: P1 + operation: server_approve + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:approve_mcp_server_submission + rationale: PROXY_ADMIN approves a pending submission into an active server +- id: mcp.server_reject.api_key.admin_only + module: mcp + tier: P2 + operation: server_reject + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:reject_mcp_server_submission + rationale: PROXY_ADMIN rejects a pending submission +- id: mcp.make_public.api_key.admin_only + module: mcp + tier: P2 + operation: make_public + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:make_mcp_servers_public + rationale: PROXY_ADMIN marks servers public for the AI Hub +- id: mcp.oauth_authorize.oauth.issues_code + module: mcp + tier: P1 + operation: oauth_authorize + auth_family: oauth + assertions: + - issues_code + source: litellm/proxy/_experimental/mcp_server/byok_oauth_endpoints.py:byok_authorize_post + rationale: Authenticated authorize stores a one-time code bound to user_id + PKCE challenge and redirects with code+state +- id: mcp.oauth_token.oauth.exchanges_code + module: mcp + tier: P1 + operation: oauth_token + auth_family: oauth + assertions: + - exchanges_code + source: litellm/proxy/_experimental/mcp_server/byok_oauth_endpoints.py:byok_token + rationale: Valid code + PKCE verifier mints a byok_session JWT and persists the per-user upstream credential +- id: mcp.oauth_token.oauth.rejects_bad_pkce + module: mcp + tier: P0 + operation: oauth_token + auth_family: oauth + assertions: + - rejects_bad_pkce + source: litellm/proxy/_experimental/mcp_server/byok_oauth_endpoints.py:byok_token + rationale: S256 verifier mismatch returns invalid_grant; the PKCE trust boundary must not be bypassable +- id: mcp.oauth_token.oauth.rejects_reused_code + module: mcp + tier: P1 + operation: oauth_token + auth_family: oauth + assertions: + - rejects_reused_code + source: litellm/proxy/_experimental/mcp_server/byok_oauth_endpoints.py:byok_token + rationale: Authorization code is one-time-use; a replay is refused +- id: mcp.oauth_metadata.none.discoverable + module: mcp + tier: P2 + operation: oauth_metadata + auth_family: none + assertions: + - discoverable + source: litellm/proxy/_experimental/mcp_server/byok_oauth_endpoints.py:oauth_authorization_server_metadata + rationale: Public .well-known AS + protected-resource metadata advertise S256 + authorization_code so MCP clients discover + the flow +- id: mcp.user_credential_set.api_key.persists + module: mcp + tier: P1 + operation: user_credential_set + auth_family: api_key + assertions: + - persists + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:store_mcp_user_credential + rationale: Caller stores its own BYOK key for a server; injected at egress on later tool calls +- id: mcp.user_credential_delete.api_key.succeeds + module: mcp + tier: P2 + operation: user_credential_delete + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:delete_mcp_user_credential + rationale: Caller revokes its own stored BYOK key +- id: mcp.oauth_user_credential_status.api_key.succeeds + module: mcp + tier: P2 + operation: oauth_user_credential_status + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:get_mcp_oauth_user_credential_status + rationale: Caller checks whether it has a stored upstream OAuth2 credential +- id: mcp.user_env_vars_set.api_key.rejects_undeclared + module: mcp + tier: P2 + operation: user_env_vars_set + auth_family: api_key + assertions: + - persists + - rejects_undeclared + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:store_mcp_user_env_vars + rationale: Per-user env vars persist, but only admin-declared var names are accepted +- id: mcp.tools_rest.api_key.succeeds + module: mcp + tier: P1 + operation: tools_rest + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:get_mcp_tools + rationale: Flat REST tool listing scoped to the calling key's allowed servers +- id: mcp.tool_search.api_key.denied_without_permission + module: mcp + tier: P1 + operation: tool_search + auth_family: api_key + assertions: + - denied_without_permission + source: litellm/proxy/_experimental/mcp_server/tool_search.py:handle_mcp_tool_search + rationale: Semantic tool-search virtual tool is gated by object_permission.mcp_tool_search_enabled +- id: mcp.tool_search.api_key.succeeds + module: mcp + tier: P2 + operation: tool_search + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/_experimental/mcp_server/tool_search.py:handle_mcp_tool_search + rationale: Semantic tool search returns ranked tools for a permitted key +- id: mcp.access_groups.api_key.succeeds + module: mcp + tier: P2 + operation: access_groups + auth_family: api_key + assertions: + - succeeds + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:get_mcp_access_groups + rationale: List MCP access groups the key can see +- id: mcp.discover.api_key.admin_only + module: mcp + tier: P2 + operation: discover + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:discover_mcp_servers + rationale: PROXY_ADMIN-only curated well-known server discovery for the UI +- id: mcp.registry_json.none.public_filtered + module: mcp + tier: P2 + operation: registry_json + auth_family: none + assertions: + - public_filtered + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:get_mcp_registry + rationale: Public registry is gated by a feature flag and IP-filtered to public servers for external callers +- id: mcp.toolset_add.api_key.admin_only + module: mcp + tier: P2 + operation: toolset_add + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:add_mcp_toolset + rationale: PROXY_ADMIN creates a toolset +- id: mcp.toolset_list.api_key.scoped + module: mcp + tier: P2 + operation: toolset_list + auth_family: api_key + assertions: + - scoped + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:fetch_mcp_toolsets + rationale: Toolset listing is filtered to the key's object_permission.mcp_toolsets +- id: mcp.toolset_delete.api_key.admin_only + module: mcp + tier: P2 + operation: toolset_delete + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:remove_mcp_toolset + rationale: PROXY_ADMIN deletes a toolset +- id: mcp.oauth_session.api_key.admin_only + module: mcp + tier: P2 + operation: oauth_session + auth_family: api_key + assertions: + - admin_only + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:add_session_mcp_server + rationale: PROXY_ADMIN caches a temp in-memory server for the OAuth setup flow (no DB write) +- id: mcp.upstream_authorize.oauth.redirects_upstream + module: mcp + tier: P2 + operation: upstream_authorize + auth_family: oauth + assertions: + - redirects_upstream + source: litellm/proxy/management_endpoints/mcp_management_endpoints.py:mcp_authorize + rationale: Gateway-managed authorization_code begins the upstream OAuth authorize for a server; the delegate-auth leg needs + a real OAuth upstream to prove end to end +- id: mcp.list_tools.api_key.team_toolset_scoped + module: mcp + tier: P0 + operation: list_tools + auth_family: api_key + assertions: + - toolset_scoped + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: 'A toolset attached to a team''s object_permission narrows the team key''s tools/list to exactly the tools it + names (LIT-5749: stored but inert before the fix)' + fail_before_fix: proven +- id: mcp.call_tool.api_key.team_toolset_denied_outside + module: mcp + tier: P0 + operation: call_tool + auth_family: api_key + assertions: + - denied_outside_toolset + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: A team key calling a tool on the granted server but outside the team's toolset is refused; the named tool stays + callable (LIT-5749) + fail_before_fix: proven +- id: mcp.list_tools.api_key.org_toolset_scoped + module: mcp + tier: P1 + operation: list_tools + auth_family: api_key + assertions: + - toolset_scoped + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: 'A toolset on an org''s object_permission caps a member team''s key that declares no MCP grants of its own (LIT-5749: + the org substitution amplifier)' + fail_before_fix: proven +- id: mcp.call_tool.api_key.org_toolset_denied_outside + module: mcp + tier: P1 + operation: call_tool + auth_family: api_key + assertions: + - denied_outside_toolset + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: An org-inherited key calling outside the org's toolset is refused while the named tool stays callable (LIT-5749) + fail_before_fix: proven +- id: mcp.list_tools.api_key.user_toolset_scoped + module: mcp + tier: P1 + operation: list_tools + auth_family: api_key + assertions: + - toolset_scoped + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: A toolset on an internal user's row ceils the servers the user's own key grants; user-level toolsets narrow, + never grant (LIT-5749) + fail_before_fix: proven +- id: mcp.call_tool.api_key.user_toolset_denied_outside + module: mcp + tier: P1 + operation: call_tool + auth_family: api_key + assertions: + - denied_outside_toolset + source: tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py:TestMcpToolsetEnforcementPerLevel + rationale: A key whose user row holds a toolset is refused calling outside it even though the key itself grants the whole + server (LIT-5749) + fail_before_fix: proven diff --git a/tests/e2e/mcp/datadog_mcp.py b/tests/e2e/mcp/datadog_mcp.py index 352b4446cfd..e16f62d7be6 100644 --- a/tests/e2e/mcp/datadog_mcp.py +++ b/tests/e2e/mcp/datadog_mcp.py @@ -1,4 +1,9 @@ -"""Shared helpers for e2e tests that register the real Datadog remote MCP server.""" +"""Shared helpers for e2e tests that register the real Datadog remote MCP server. + +The search_datadog_logs arguments these suites send (query, from, to, max_tokens) come from +https://docs.datadoghq.com/mcp_server/tools/ (checked 2026-09-19). McpToolEntry.assert_arguments_are_documented +compares them with the schema the gateway advertises live, so a failure there after a Datadog rename is stale, +not broken litellm code""" from __future__ import annotations diff --git a/tests/e2e/mcp/mcp_client.py b/tests/e2e/mcp/mcp_client.py index 56f7fffba29..71e2ad43c67 100644 --- a/tests/e2e/mcp/mcp_client.py +++ b/tests/e2e/mcp/mcp_client.py @@ -65,10 +65,22 @@ class McpToolMcpInfo(BaseModel): alias: str | None = None +class McpToolInputSchema(BaseModel): + properties: dict[str, object] = {} + + class McpToolEntry(BaseModel): name: str description: str | None = None mcp_info: McpToolMcpInfo | None = None + input_schema: McpToolInputSchema = Field(alias="inputSchema") + + def assert_arguments_are_documented(self, arguments: McpToolArguments) -> None: + undocumented = frozenset(arguments).difference(self.input_schema.properties) + assert not undocumented, ( + f"tool {self.name!r} arguments absent from advertised input schema: {sorted(undocumented)}; " + f"documented arguments: {sorted(self.input_schema.properties)}" + ) class McpToolsListResponse(BaseModel): @@ -78,18 +90,20 @@ class McpToolsListResponse(BaseModel): def tool_names_for_server(self, server_id: str) -> frozenset[str]: return frozenset( - tool.name - for tool in self.tools - if tool.mcp_info is not None and tool.mcp_info.server_id == server_id + tool.name for tool in self.tools if tool.mcp_info is not None and tool.mcp_info.server_id == server_id ) def tool_name_containing(self, server_id: str, needle: str) -> str | None: + tool = self.tool_containing(server_id, needle) + return tool.name if tool is not None else None + + def tool_containing(self, server_id: str, needle: str) -> McpToolEntry | None: needle_l = needle.lower() for tool in self.tools: if tool.mcp_info is None or tool.mcp_info.server_id != server_id: continue if needle_l in tool.name.lower() or tool.name.lower().endswith(needle_l): - return tool.name + return tool return None @@ -224,9 +238,7 @@ class McpClient: McpServerListResponse, settled=lambda response: any(row.server_id == server_id for row in response.root), ) - return next( - row for response in registered.values() for row in response.root if row.server_id == server_id - ) + return next(row for response in registered.values() for row in response.root if row.server_id == server_id) def generate_key( self, @@ -263,8 +275,16 @@ class McpClient: ) def await_tool(self, key: str, server_id: str, needle: str) -> str: + return self.await_tool_entry(key, server_id, needle).name + + def await_tool_entry(self, key: str, server_id: str, needle: str) -> McpToolEntry: + tool = self.await_tool_catalog(key, server_id, needle).tool_containing(server_id, needle) + assert tool is not None + return tool + + def await_tool_catalog(self, key: str, server_id: str, needle: str) -> McpToolsListResponse: """Poll tools/list until `server_id` serves a tool matching `needle`, and - return its fully-qualified name. Fails at poll_timeout. + return that catalog snapshot. Fails at poll_timeout. /v1/mcp/server returns as soon as the DB row is written, but the gateway runs the initialize + tools/list handshake against the upstream lazily on @@ -276,9 +296,9 @@ class McpClient: while True: result = self.list_tools(key) if isinstance(result, Success): - tool_name = result.data.tool_name_containing(server_id, needle) - if tool_name is not None: - return tool_name + tool = result.data.tool_containing(server_id, needle) + if tool is not None: + return result.data if time.monotonic() >= deadline: raise AssertionError( f"server {server_id} never served a tool matching {needle!r} within " @@ -345,13 +365,10 @@ class McpClient: if isinstance(last, UnknownApiError) and last.status_code == 403: return last if not _is_mcp_not_synced(last, tool_name=name): - raise AssertionError( - f"ungranted key's tools/call was not 403 access_denied: {last}" - ) + raise AssertionError(f"ungranted key's tools/call was not 403 access_denied: {last}") if time.monotonic() >= deadline: raise AssertionError( - f"ungranted key never got 403 for {name!r} within {self.proxy.poll_timeout}s; " - f"last result: {last}" + f"ungranted key never got 403 for {name!r} within {self.proxy.poll_timeout}s; last result: {last}" ) time.sleep(self.proxy.poll_interval) @@ -397,9 +414,7 @@ class McpClient: return self.proxy.transport.post( "/mcp-rest/tools/call", headers=ApiKeyHeaders(x_litellm_api_key=key), - json=McpCallToolBody( - name=name, arguments=dict(arguments), server_id=server_id - ), + json=McpCallToolBody(name=name, arguments=dict(arguments), server_id=server_id), response_type=McpCallToolResponse, ) @@ -432,9 +447,7 @@ def _is_mcp_not_synced( # Gateway: "Tool search_datadog_logs not found" (optionally inside a longer message) if tool_name is not None: - return ( - re.search(rf"\btool\s+{re.escape(tool_name)}\s+not found\b", body_l) is not None - ) + return re.search(rf"\btool\s+{re.escape(tool_name)}\s+not found\b", body_l) is not None return re.search(r"\btool\s+\S+\s+not found\b", body_l) is not None diff --git a/tests/e2e/mcp/test_mcp_datadog_e2e.py b/tests/e2e/mcp/test_mcp_datadog_e2e.py index 031fbf6d936..821e48a8783 100644 --- a/tests/e2e/mcp/test_mcp_datadog_e2e.py +++ b/tests/e2e/mcp/test_mcp_datadog_e2e.py @@ -34,8 +34,7 @@ def _assert_datadog_logger_active(proxy: ProxyClient) -> None: f"/health/readiness/details must answer 200, got {result.status_code}: {result.body[:300]}" ) assert DD_LOGGER_NAME in result.body, ( - f"the proxy must report the {DD_LOGGER_NAME} callback active " - f"(callbacks + DD_* env); got: {result.body[:400]}" + f"the proxy must report the {DD_LOGGER_NAME} callback active (callbacks + DD_* env); got: {result.body[:400]}" ) @@ -78,21 +77,12 @@ class TestDatadogMcpRoundTrip: "within the poll deadline; MCP search would have nothing to find" ) - tool_name = client.await_tool(key, server_id, SEARCH_LOGS_TOOL) - call = client.await_call_tool( - key, - server_id=server_id, - name=tool_name, - arguments={ - "query": marker, - "from": DD_SEARCH_FROM, - "to": "now", - "max_tokens": 5000, - }, - ) + tool = client.await_tool_entry(key, server_id, SEARCH_LOGS_TOOL) + arguments = {"query": marker, "from": DD_SEARCH_FROM, "to": "now", "max_tokens": 5000} + tool.assert_arguments_are_documented(arguments) + call = client.await_call_tool(key, server_id=server_id, name=tool.name, arguments=arguments) assert call.is_error is not True, f"search_datadog_logs errored: {call}" body = call.all_text assert marker in body, ( - f"search_datadog_logs response must include the seeded marker {marker!r}; " - f"got: {body[:800]!r}" + f"search_datadog_logs response must include the seeded marker {marker!r}; got: {body[:800]!r}" ) diff --git a/tests/e2e/mcp/test_mcp_guardrail_e2e.py b/tests/e2e/mcp/test_mcp_guardrail_e2e.py index 92c632cb316..683530db0cb 100644 --- a/tests/e2e/mcp/test_mcp_guardrail_e2e.py +++ b/tests/e2e/mcp/test_mcp_guardrail_e2e.py @@ -89,9 +89,7 @@ class TestMcpToolCallGuardrail: marker = unique_marker() banned_keyword = f"e2eblocked{marker}" - guardrail_id = client.register_mcp_content_filter( - name=f"e2e-mcp-cf-{marker}", blocked_keyword=banned_keyword - ) + guardrail_id = client.register_mcp_content_filter(name=f"e2e-mcp-cf-{marker}", blocked_keyword=banned_keyword) guardrail_created_at = time.monotonic() resources.defer(lambda: client.delete_guardrail(guardrail_id)) @@ -100,7 +98,7 @@ class TestMcpToolCallGuardrail: key = client.generate_key(user_id=f"e2e-mcp-guard-{marker}", mcp_servers=[server_id]) resources.defer(lambda: client.proxy.delete_key(key)) - tool_name = client.await_tool(key, server_id, SEARCH_LOGS_TOOL) + tool = client.await_tool_entry(key, server_id, SEARCH_LOGS_TOOL) def search(query: str) -> Result[McpCallToolResponse]: arguments: McpToolArguments = { @@ -109,7 +107,8 @@ class TestMcpToolCallGuardrail: "to": "now", "max_tokens": 500, } - return client.call_tool(key, server_id=server_id, name=tool_name, arguments=arguments) + tool.assert_arguments_are_documented(arguments) + return client.call_tool(key, server_id=server_id, name=tool.name, arguments=arguments) # Registering the guardrail is a control-plane write; the data-plane worker # that serves tools/call picks it up on its next guardrail sync, so an @@ -162,6 +161,4 @@ class TestMcpToolCallGuardrail: f"a clean MCP tool call must reach the server and not error, got: {result}" ) case _: - pytest.fail( - f"a clean MCP tool call must pass the guardrail and reach the server; got {allowed}" - ) + pytest.fail(f"a clean MCP tool call must pass the guardrail and reach the server; got {allowed}") diff --git a/tests/e2e/mcp/test_mcp_key_access_e2e.py b/tests/e2e/mcp/test_mcp_key_access_e2e.py index c00d67bc9cf..7b63423b180 100644 --- a/tests/e2e/mcp/test_mcp_key_access_e2e.py +++ b/tests/e2e/mcp/test_mcp_key_access_e2e.py @@ -77,8 +77,7 @@ class TestMcpKeyWithoutAccessIsDenied: denied_tools = unwrap(client.list_tools(denied_key)).tool_names_for_server(server_id) assert denied_tools == frozenset(), ( - f"ungranted key saw the server's tools; tools/list leaked across the permission " - f"boundary: {denied_tools}" + f"ungranted key saw the server's tools; tools/list leaked across the permission boundary: {denied_tools}" ) @pytest.mark.covers("mcp.call_tool.api_key.denied_without_permission") @@ -93,7 +92,7 @@ class TestMcpKeyWithoutAccessIsDenied: permitted_key = _key(client, resources, mcp_servers=[server_id]) denied_key = _key(client, resources, mcp_servers=None) - tool_name = client.await_tool(permitted_key, server_id, SEARCH_LOGS_TOOL) + tool = client.await_tool_entry(permitted_key, server_id, SEARCH_LOGS_TOOL) search_args = { "query": "service:litellm", @@ -101,14 +100,13 @@ class TestMcpKeyWithoutAccessIsDenied: "to": "now", "max_tokens": 1000, } + tool.assert_arguments_are_documented(search_args) permitted_call = client.await_call_tool( - permitted_key, server_id=server_id, name=tool_name, arguments=search_args + permitted_key, server_id=server_id, name=tool.name, arguments=search_args ) assert permitted_call.is_error is not True, f"granted key's tool call errored: {permitted_call}" - denied = client.await_call_tool_denied( - denied_key, server_id=server_id, name=tool_name, arguments=search_args - ) + denied = client.await_call_tool_denied(denied_key, server_id=server_id, name=tool.name, arguments=search_args) assert "access_denied" in denied.body, f"403 was not an MCP access denial: {denied.body}" @@ -126,17 +124,21 @@ class TestMcpHealthVisibility: permitted: Final = _key(client, resources, mcp_servers=[server_x]) tool: Final = client.await_tool(permitted, server_x, SEARCH_LOGS_TOOL) result: Final = client.await_call_tool( - permitted, server_id=server_x, name=tool, + permitted, + server_id=server_x, + name=tool, arguments={"query": "service:litellm", "from": DD_SEARCH_FROM, "to": "now", "max_tokens": 1000}, ) assert result.is_error is not True, f"permitted control failed: {result}" for grants in ([server_x], [server_y], []): - key = client.proxy.generate_key(KeyGenerateBody( - user_id=f"e2e-mcp-health-{unique_marker()}", - allowed_routes=["/v1/mcp/server", "/v1/mcp/server/health"], - object_permission=ObjectPermission(mcp_servers=grants), - )) + key = client.proxy.generate_key( + KeyGenerateBody( + user_id=f"e2e-mcp-health-{unique_marker()}", + allowed_routes=["/v1/mcp/server", "/v1/mcp/server/health"], + object_permission=ObjectPermission(mcp_servers=grants), + ) + ) resources.defer(lambda key=key: client.proxy.delete_key(key)) listed = unwrap(client.list_servers(key)).root assert {row.server_id for row in listed}.intersection(owned) == set(grants) diff --git a/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py b/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py index 6b901145eb1..135830c52ed 100644 --- a/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py +++ b/tests/e2e/mcp/test_mcp_toolset_enforcement_e2e.py @@ -12,15 +12,25 @@ upstream). from __future__ import annotations +from dataclasses import replace from typing import Final import pytest from datadog_mcp import SEARCH_LOGS_TOOL, register_datadog_mcp -from e2e_config import unique_marker -from e2e_http import unwrap +from e2e_config import DD_SEARCH_FROM, unique_marker +from e2e_http import UnknownApiError, unwrap from lifecycle import ResourceManager from mcp_client import McpClient -from models import ToolsetCreateBody, ToolsetTool +from management.management_client import build_client as build_management_client +from models import ( + KeyGenerateBody, + ObjectPermission, + OrgNewBody, + TeamNewBody, + ToolsetCreateBody, + ToolsetTool, + UserNewBody, +) pytestmark = pytest.mark.e2e @@ -65,8 +75,10 @@ class TestMcpToolsetEnforcement: client.await_registered(server_id) catalog_key: Final = _key(client, resources, "catalog", server_id=server_id) - known_wire: Final = client.await_tool(catalog_key, server_id, SEARCH_LOGS_TOOL) - catalog: Final = unwrap(client.list_tools(catalog_key)).tool_names_for_server(server_id) + snapshot: Final = client.await_tool_catalog(catalog_key, server_id, SEARCH_LOGS_TOOL) + known_wire: Final = snapshot.tool_name_containing(server_id, SEARCH_LOGS_TOOL) + assert known_wire is not None + catalog: Final = snapshot.tool_names_for_server(server_id) assert len(catalog) > 2, ( f"the Datadog core toolset must serve more tools than the toolset names, or the " f"restriction has nothing to hide; got {sorted(catalog)}" @@ -93,3 +105,82 @@ class TestMcpToolsetEnforcement: f"a key granted the toolset must list exactly its two tools; " f"got {sorted(listed)}, expected {sorted(chosen_wire)}" ) + + +def _assert_principal_toolset(client: McpClient, resources: ResourceManager, principal: str) -> None: + server_id = register_datadog_mcp(client, resources, allowed_tools=None) + client.await_registered(server_id) + control_key = _key(client, resources, "principal-control", server_id=server_id) + toolset = client.proxy.create_toolset( + ToolsetCreateBody( + toolset_name=f"e2e_principal_{unique_marker()}", + tools=[ToolsetTool(server_id=server_id, tool_name=SEARCH_LOGS_TOOL)], + ) + ) + resources.defer(lambda: client.proxy.delete_toolset(toolset.toolset_id)) + assert [(tool.server_id, tool.tool_name) for tool in toolset.tools] == [(server_id, SEARCH_LOGS_TOOL)] + management = build_management_client(client.proxy) + permission = ObjectPermission(mcp_servers=[server_id], mcp_toolsets=[toolset.toolset_id]) + if principal == "user": + user_id = management.create_user( + UserNewBody( + user_email=f"e2e-toolset-{unique_marker()}@example.com", + user_role="internal_user", + auto_create_key=False, + object_permission=ObjectPermission(mcp_toolsets=[toolset.toolset_id]), + ) + ) + resources.defer(lambda: management.delete_user(user_id)) + scoped_key = client.generate_key(user_id=user_id, mcp_servers=[server_id]) + else: + if principal == "organization": + org_id = management.create_org( + OrgNewBody( + organization_alias=f"e2e-toolset-{unique_marker()}", + object_permission=permission, + ) + ) + resources.defer(lambda: management.delete_org(org_id)) + team_body = TeamNewBody(team_alias=f"e2e-toolset-{unique_marker()}", organization_id=org_id) + else: + team_body = TeamNewBody(team_alias=f"e2e-toolset-{unique_marker()}", object_permission=permission) + team_id = management.create_team(team_body) + resources.defer(lambda: management.delete_team(team_id)) + scoped_key = client.proxy.generate_key(KeyGenerateBody(team_id=team_id)) + resources.defer(lambda: client.proxy.delete_key(scoped_key)) + + for transport in client.proxy.replicas_for("/mcp-rest/tools/list").values(): + replica = McpClient(proxy=replace(client.proxy, transport=transport)) + snapshot = replica.await_tool_catalog(control_key, server_id, SEARCH_LOGS_TOOL) + granted = snapshot.tool_containing(server_id, SEARCH_LOGS_TOOL) + assert granted is not None + catalog = snapshot.tool_names_for_server(server_id) + outside = sorted(catalog - {granted.name}) + assert outside, f"uncapped upstream must expose a tool outside the grant: {catalog}" + expected = frozenset({granted.name}) + assert replica.await_tools(scoped_key, server_id, expected=expected) == expected + denied = replica.call_tool(scoped_key, server_id=server_id, name=outside[0], arguments={}) + assert isinstance(denied, UnknownApiError) and denied.status_code == 403, denied + assert "is not allowed for your key/team on server" in denied.body, denied.body + arguments = {"query": f"service:e2e-toolset-{unique_marker()}", "from": DD_SEARCH_FROM} + granted.assert_arguments_are_documented(arguments) + result = unwrap(replica.call_tool(scoped_key, server_id=server_id, name=granted.name, arguments=arguments)) + assert result.is_error is not True, result.all_text + + +class TestMcpToolsetEnforcementPerLevel: + @pytest.mark.covers( + "mcp.list_tools.api_key.team_toolset_scoped", "mcp.call_tool.api_key.team_toolset_denied_outside" + ) + def test_team_toolset_narrows_team_key(self, client: McpClient, resources: ResourceManager) -> None: + _assert_principal_toolset(client, resources, "team") + + @pytest.mark.covers("mcp.list_tools.api_key.org_toolset_scoped", "mcp.call_tool.api_key.org_toolset_denied_outside") + def test_org_toolset_caps_inherited_team_key(self, client: McpClient, resources: ResourceManager) -> None: + _assert_principal_toolset(client, resources, "organization") + + @pytest.mark.covers( + "mcp.list_tools.api_key.user_toolset_scoped", "mcp.call_tool.api_key.user_toolset_denied_outside" + ) + def test_user_toolset_ceils_own_key_grant(self, client: McpClient, resources: ResourceManager) -> None: + _assert_principal_toolset(client, resources, "user") diff --git a/tests/e2e/models.py b/tests/e2e/models.py index 6e66529ec8f..0e1420208af 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -1437,6 +1437,7 @@ class TeamMetadata(BaseModel): class TeamNewBody(BaseModel): + object_permission: ObjectPermission | None = None team_alias: str models: list[str] = [] team_id: str | None = None @@ -1499,6 +1500,7 @@ UserRole = Literal["proxy_admin", "proxy_admin_viewer", "internal_user", "intern class UserNewBody(BaseModel): + object_permission: ObjectPermission | None = None user_email: str user_role: UserRole user_id: str | None = None @@ -1551,6 +1553,7 @@ class UserListResponse(BaseModel): class OrgNewBody(BaseModel): + object_permission: ObjectPermission | None = None organization_alias: str models: list[str] = []