From d06b38c26880aa9a4effcdaa5d879444956f697d Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 11:28:22 -0700 Subject: [PATCH] test(mcp): add pinned official conformance smoke checks --- .circleci/config.yml | 11 ++ .circleci/scripts/install_mcp_conformance.sh | 35 +++++ .circleci/scripts/run_integration.sh | 1 + tests/integration/_support/conformance.py | 143 ++++++++++++++++++ .../mcp/test_mcp_official_conformance.py | 28 ++++ .../integration_support/test_conformance.py | 99 ++++++++++++ 6 files changed, 317 insertions(+) create mode 100644 .circleci/scripts/install_mcp_conformance.sh create mode 100644 tests/integration/_support/conformance.py create mode 100644 tests/integration/mcp/test_mcp_official_conformance.py create mode 100644 tests/unit/integration_support/test_conformance.py diff --git a/.circleci/config.yml b/.circleci/config.yml index 7d4e2e40769..f11ce44c893 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -3241,6 +3241,17 @@ jobs: - run: name: Build the candidate dashboard command: cd ui/litellm-dashboard && NEXT_TELEMETRY_DISABLED=1 npm run build + - when: + condition: + equal: [mcp, << parameters.suite >>] + steps: + - install_node + - run: + name: Install pinned official MCP conformance runner and reference + command: | + export MCP_CONFORMANCE_ROOT="/tmp/litellm-mcp-conformance" + bash .circleci/scripts/install_mcp_conformance.sh + echo 'export MCP_CONFORMANCE_ROOT="/tmp/litellm-mcp-conformance"' >> "$BASH_ENV" - start_postgres: image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5 server_args: "-c shared_preload_libraries=pg_stat_statements -c pg_stat_statements.track=all -c pg_stat_statements.max=20000" diff --git a/.circleci/scripts/install_mcp_conformance.sh b/.circleci/scripts/install_mcp_conformance.sh new file mode 100644 index 00000000000..8bc5eb22750 --- /dev/null +++ b/.circleci/scripts/install_mcp_conformance.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +set -euo pipefail + +root="${MCP_CONFORMANCE_ROOT:?Set MCP_CONFORMANCE_ROOT to a fresh directory}" +test ! -e "$root" +archive="$(mktemp)" +trap 'rm -f "$archive"' EXIT +curl --fail --location --silent --show-error \ + 'https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/7169291ec0b68eb370fddcd9947313ab0d5e4156' \ + --output "$archive" +PYTHONPATH=tests .venv/bin/python - "$archive" <<'PY' +import sys +from pathlib import Path +from integration._support.conformance import verify_archive + +verify_archive(Path(sys.argv[1]), "51c1e27027f36be5b5f067746eb3a0f240fc3fbb7adcbd64bc4a79d86db65cd8") +PY +mkdir -p "$root" +tar -xzf "$archive" --strip-components=1 -C "$root" +npm ci --ignore-scripts --prefix "$root" +# The current reference misclassifies legacy initialize _meta as stateless traffic. +# Use the last official legacy reference; leave its source and lockfile untouched. +curl --fail --location --silent --show-error \ + 'https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/8f3994c75ff1aed1e39f91cff9358e2bc2c81dcd' \ + --output "$archive" +PYTHONPATH=tests .venv/bin/python - "$archive" <<'PYREF' +import sys +from pathlib import Path +from integration._support.conformance import verify_archive + +verify_archive(Path(sys.argv[1]), "181119bd222f29db208b2952537430fcc865f91b64f2ac3d7525a67f490ebaa1") +PYREF +mkdir "$root/legacy-reference" +tar -xzf "$archive" --strip-components=1 -C "$root/legacy-reference" +npm ci --ignore-scripts --prefix "$root/legacy-reference/examples/servers/typescript" diff --git a/.circleci/scripts/run_integration.sh b/.circleci/scripts/run_integration.sh index 47ad2274e2f..4b724cc5ad9 100644 --- a/.circleci/scripts/run_integration.sh +++ b/.circleci/scripts/run_integration.sh @@ -234,6 +234,7 @@ env -i PATH="$PATH" HOME="$HOME" PYTHONPATH="$PYTHONPATH" \ INTEGRATION_PROXY_DATABASE_URL="$INTEGRATION_PROXY_DATABASE_URL" \ INTEGRATION_PROXY_READ_REPLICA_URL="$INTEGRATION_PROXY_READ_REPLICA_URL" \ INTEGRATION_ROUTING="$INTEGRATION_ROUTING" \ + MCP_CONFORMANCE_ROOT="${MCP_CONFORMANCE_ROOT:-}" \ .venv/bin/python tests/integration/run.py "$suite" --results "$results" "${node_files[@]}" if [ "${INTEGRATION_COVERAGE:-0}" = 1 ]; then diff --git a/tests/integration/_support/conformance.py b/tests/integration/_support/conformance.py new file mode 100644 index 00000000000..82317850081 --- /dev/null +++ b/tests/integration/_support/conformance.py @@ -0,0 +1,143 @@ +import hashlib +import os +import queue +import signal +import subprocess +from collections.abc import Iterator +from contextlib import contextmanager +from pathlib import Path +from typing import TYPE_CHECKING, Final, Literal + +import httpx +from pydantic import BaseModel, Field, JsonValue, TypeAdapter + +if TYPE_CHECKING: + from integration._support.mcp import McpPeer + + +class ConformanceCheck(BaseModel): + id: str + status: Literal["SUCCESS", "FAILURE", "WARNING", "INFO"] + errorMessage: str | None = None + details: dict[str, JsonValue] = Field(default_factory=dict) + + +def verify_archive(archive: Path, expected_sha256: str) -> None: + actual: Final = hashlib.sha256(archive.read_bytes()).hexdigest() + assert actual == expected_sha256, f"SHA-256 mismatch for {archive.name}: {actual}" + + +def read_checks(directory: Path, scenario: str) -> tuple[ConformanceCheck, ...]: + reports: Final = tuple(directory.rglob("checks.json")) + assert len(reports) == 1, f"conformance {scenario}: expected one fresh report, found {len(reports)}" + checks: Final = TypeAdapter(tuple[ConformanceCheck, ...]).validate_json(reports[0].read_bytes()) + assert len({check.id for check in checks}) == len(checks), f"conformance {scenario}: duplicate checks" + assert tuple(check.status for check in checks if check.id == scenario) == ("SUCCESS",), ( + f"conformance {scenario}: required scenario did not pass: {checks}" + ) + assert all(check.status != "FAILURE" for check in checks), f"conformance {scenario}: failed checks: {checks}" + if scenario == "tools-call-simple-text": + result: Final = next(check for check in checks if check.id == scenario).details.get("result") + # The official scenario accepts any nonempty text, including tool errors. + assert isinstance(result, dict) and result.get("isError", False) is False, f"reference payload: {result}" + assert result.get("content") == [{"type": "text", "text": "This is a simple text response for testing."}], ( + f"reference payload: {result}" + ) + return checks + + +def run_scenario(root: Path, url: str, scenario: str, directory: Path) -> tuple[ConformanceCheck, ...]: + directory.mkdir(parents=True, exist_ok=False) + with (directory / "runner.log").open("w") as log: + result: Final = subprocess.run( + [ + "node", + "--import", + "tsx", + "src/index.ts", + "server", + "--url", + url, + "--scenario", + scenario, + "--spec-version", + "2025-11-25", + "--output-dir", + str(directory.resolve()), + "--timeout", + "30000", + ], + cwd=root, + stdout=log, + stderr=subprocess.STDOUT, + timeout=45, + check=False, + ) + assert result.returncode == 0, (directory / "runner.log").read_text() + return read_checks(directory, scenario) + + +@contextmanager +def reference_server(root: Path, directory: Path, port: int) -> Iterator["McpPeer"]: + from integration._support.client import eventually + from integration._support.mcp import McpPeer + from integration._support.process import signal_group, stop_root_process + + directory.mkdir(parents=True, exist_ok=True) + url: Final = f"http://127.0.0.1:{port}" + with (directory / "reference.log").open("w") as log: + process: Final = subprocess.Popen( + ["node", "--import", "tsx", "everything-server.ts"], + cwd=root / "legacy-reference/examples/servers/typescript", + env={"PATH": os.environ["PATH"], "PORT": str(port)}, + stdout=log, + stderr=subprocess.STDOUT, + start_new_session=True, + ) + try: + with httpx.Client(timeout=1, trust_env=False) as client: + + def ready() -> bool: + assert process.poll() is None, (directory / "reference.log").read_text() + try: + return client.get(url + "/mcp").status_code == 400 + except httpx.TransportError: + return False + + eventually(ready, bool, seconds=15) + yield McpPeer(url + "/mcp", queue.Queue()) + finally: + stopped: Final = stop_root_process(process) + if not stopped: + signal_group(process.pid, signal.SIGKILL) + process.wait(timeout=3) + assert stopped, "Official reference required forced cleanup" + + +@contextmanager +def authenticated_endpoint(target: str, key: str) -> Iterator[str]: + from integration._support.asgi import asgi_server + from starlette.applications import Starlette + from starlette.requests import Request + from starlette.responses import StreamingResponse + from starlette.routing import Mount + from starlette.types import Receive, Scope, Send + + async def forward(scope: Scope, receive: Receive, send: Send) -> None: + request: Final = Request(scope, receive) + headers: Final = tuple( + (name, value) for name, value in request.headers.raw if name.lower() != b"authorization" + ) + ((b"authorization", f"Bearer {key}".encode()),) + body: Final = await request.body() + async with httpx.AsyncClient(timeout=35, trust_env=False) as client: + async with client.stream(request.method, target, headers=headers, content=body) as response: + streamed: Final = StreamingResponse(response.aiter_raw(), status_code=response.status_code) + streamed.raw_headers = [ + (name, value) + for name, value in response.headers.raw + if name.lower() not in (b"transfer-encoding", b"connection") + ] + await streamed(scope, receive, send) + + with asgi_server(Starlette(routes=[Mount("/mcp", app=forward)])) as url: + yield url + "/mcp/" diff --git a/tests/integration/mcp/test_mcp_official_conformance.py b/tests/integration/mcp/test_mcp_official_conformance.py new file mode 100644 index 00000000000..0877fc9457e --- /dev/null +++ b/tests/integration/mcp/test_mcp_official_conformance.py @@ -0,0 +1,28 @@ +import os +import uuid +from pathlib import Path +from typing import Final + +import pytest +from integration._support.client import Gateway +from integration._support.conformance import authenticated_endpoint, reference_server, run_scenario +from integration._support.mcp import official_client_outcomes, register_mcp + + +@pytest.mark.parametrize("name", ("server-initialize", "tools-list", "tools-call-image")) +def test_official_scenario_through_gateway(gateway: Gateway, tmp_path: Path, unused_tcp_port: int, name: str) -> None: + root: Final = Path(os.environ["MCP_CONFORMANCE_ROOT"]) + output: Final = Path(os.environ.get("INTEGRATION_RESULTS_DIR", str(tmp_path))) / f"conformance-{uuid.uuid4().hex}" + with reference_server(root, output, unused_tcp_port) as reference, gateway.scenario() as scenario: + alias: Final = "official" + uuid.uuid4().hex[:8] + identity: Final = register_mcp(scenario, reference, alias, mcp_info={"protocol_version": "2025-11-25"}) + key: Final = scenario.key(object_permission={"mcp_servers": [identity]}) + direct: Final = run_scenario(root, reference.url, name, output / "direct") + endpoint: Final = str(gateway.client.base_url).rstrip("/") + f"/{alias}/mcp" + with authenticated_endpoint(endpoint, key) as authenticated: + proxied: Final = run_scenario(root, authenticated, name, output / "gateway") + assert tuple(check.status for check in direct if check.id == name) == ("SUCCESS",) + assert tuple(check.status for check in proxied if check.id == name) == ("SUCCESS",) + listed, called = official_client_outcomes(gateway, key, f"/{alias}/mcp", "test_simple_text", {}) + assert f"{alias}-test_simple_text" in listed.tools, listed + assert called.ok and called.text == "This is a simple text response for testing.", called diff --git a/tests/unit/integration_support/test_conformance.py b/tests/unit/integration_support/test_conformance.py new file mode 100644 index 00000000000..c7b0795479f --- /dev/null +++ b/tests/unit/integration_support/test_conformance.py @@ -0,0 +1,99 @@ +import hashlib +import json +from pathlib import Path +from typing import Final + +import pytest +from tests.integration._support.conformance import read_checks, verify_archive +from pydantic import ValidationError + + +def test_changed_archive_is_rejected(tmp_path: Path) -> None: + archive: Final = tmp_path / "reference.tar.gz" + archive.write_bytes(b"changed reference") + with pytest.raises(AssertionError, match="SHA-256"): + verify_archive(archive, hashlib.sha256(b"approved reference").hexdigest()) + + +def test_matching_archive_is_accepted(tmp_path: Path) -> None: + archive: Final = tmp_path / "reference.tar.gz" + archive.write_bytes(b"approved reference") + assert verify_archive(archive, "5a870d3ee9520e3a912c735d007d7b40f6d3cbf8b61d909ebb8361f423d4ba1e") is None + + +@pytest.mark.parametrize( + "checks", + ( + [], + [{"id": "tools-list", "status": "FAILURE"}], + [{"id": "tools-list", "status": "WARNING"}], + [{"id": "wire-schema", "status": "SUCCESS"}], + [{"id": "tools-list", "status": "SUCCESS"}, {"id": "wire-schema", "status": "FAILURE"}], + [{"id": "tools-list", "status": "SUCCESS"}, {"id": "tools-list", "status": "SUCCESS"}], + ), +) +def test_incomplete_or_failed_scenario_cannot_pass(tmp_path: Path, checks: list[dict[str, str]]) -> None: + (tmp_path / "checks.json").write_text(json.dumps(checks)) + with pytest.raises(AssertionError, match="conformance"): + read_checks(tmp_path, "tools-list") + + +def test_skipped_scenario_without_a_report_cannot_pass(tmp_path: Path) -> None: + with pytest.raises(AssertionError, match="conformance"): + read_checks(tmp_path, "tools-list") + + +def test_multiple_reports_cannot_supply_a_stale_pass(tmp_path: Path) -> None: + for name in ("previous", "current"): + directory: Final = tmp_path / name + directory.mkdir() + (directory / "checks.json").write_text('[{"id":"tools-list","status":"SUCCESS"}]') + with pytest.raises(AssertionError, match="conformance"): + read_checks(tmp_path, "tools-list") + + +def test_passing_scenario_keeps_nonbinding_diagnostics(tmp_path: Path) -> None: + (tmp_path / "checks.json").write_text('[{"id":"tools-list","status":"SUCCESS"},{"id":"advisory","status":"INFO"}]') + checks: Final = read_checks(tmp_path, "tools-list") + assert tuple((check.id, check.status) for check in checks) == (("tools-list", "SUCCESS"), ("advisory", "INFO")) + + +def test_unrecognized_result_status_cannot_pass(tmp_path: Path) -> None: + (tmp_path / "checks.json").write_text('[{"id":"tools-list","status":"SKIPPED"}]') + with pytest.raises(ValidationError, match="status"): + read_checks(tmp_path, "tools-list") + + +@pytest.mark.parametrize( + "result", + ( + {"content": [{"type": "text", "text": "Tool not found"}], "isError": True}, + {"content": [{"type": "text", "text": "wrong upstream payload"}]}, + ), +) +def test_official_text_success_cannot_hide_an_error_or_wrong_payload(tmp_path: Path, result: dict[str, object]) -> None: + (tmp_path / "checks.json").write_text( + json.dumps([{"id": "tools-call-simple-text", "status": "SUCCESS", "details": {"result": result}}]) + ) + with pytest.raises(AssertionError, match="reference payload"): + read_checks(tmp_path, "tools-call-simple-text") + + +def test_official_text_success_requires_the_reference_payload(tmp_path: Path) -> None: + (tmp_path / "checks.json").write_text( + json.dumps( + [ + { + "id": "tools-call-simple-text", + "status": "SUCCESS", + "details": { + "result": { + "content": [{"type": "text", "text": "This is a simple text response for testing."}], + "isError": False, + } + }, + } + ] + ) + ) + assert read_checks(tmp_path, "tools-call-simple-text")[0].status == "SUCCESS"