From fd23eacacbaab74788be1a2068088495d2e3f182 Mon Sep 17 00:00:00 2001 From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:57:07 -0700 Subject: [PATCH] test(mcp): protect the conformance baseline from silent gaps --- tests/integration/README.md | 52 ++++++- tests/integration/_support/conformance.py | 94 ++++++++---- .../integration/mcp/conformance_baseline.json | 72 ++++++++++ .../integration_support/test_conformance.py | 136 ++++++++++++++++++ 4 files changed, 321 insertions(+), 33 deletions(-) create mode 100644 tests/integration/mcp/conformance_baseline.json diff --git a/tests/integration/README.md b/tests/integration/README.md index 277df064a95..74ac09a08ad 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -14,7 +14,7 @@ Reuse the existing canned provider handlers through `_support/upstream.py`. It r The CircleCI workflow starts its own database and Redis, restricts test-phase egress to its owned services and writes JUnit plus an executed-node manifest. Missing setup, failed cleanup or a selected test with neither a passed call nor a skip fail qualification. Skipped nodes are listed under `skipped` in `execution.json`, so the skip reasons double as the open bug list. Existing GitHub Actions jobs do not own these tests -There is no per-node manifest. The runner fails only when pytest fails, when collection errors, or when a selected file collects zero tests. Older tests still carry `@pytest.mark.covers(...)` decorators; the marker stays registered so they collect, but the IDs are not checked against anything and new tests should not use it. The GitHub Actions coverage census reads the `GROUPS` literal in `run.py` and treats every `tests/integration//test_*.py` file in a scheduled group as owned by CircleCI +Groups other than MCP have no per-node manifest. Those groups fail when pytest fails, when collection errors, or when a selected file collects zero tests. MCP additionally enforces the required baseline described below. Older tests still carry `@pytest.mark.covers(...)` decorators; the marker stays registered so they collect, but the IDs are not checked against anything and new tests should not use it. The GitHub Actions coverage census reads the `GROUPS` literal in `run.py` and treats every `tests/integration//test_*.py` file in a scheduled group as owned by CircleCI Provider sentinels currently use the controlled server, not live recordings. The provider shard also runs the existing strict replay controls for changed requests, exhausted interactions, leftover interactions and no provider connection. Future recorded scenarios must use that replay-only implementation; missing recordings cannot fall back to a real provider. The observation endpoint is destructive and the current selection runs serially against one owned upstream @@ -39,7 +39,11 @@ Browser contracts live in `tests/e2e/ui/tests/integrationCritical` and run only Two always-on `-replica` CircleCI jobs (management, database) run their groups in replica mode, where every proxy connects through a real `litellm_writer` role and a real read-only `litellm_reader` role against the same PostgreSQL. Nothing is captured there: the job passes when the tests pass, and a write routed to the read-only reader fails the test that issued it. A deeper check runs on demand as the `routing_parity` workflow, triggered through the CircleCI API v2 pipeline endpoint on the PR branch with `{"parameters": {"routing_parity_base": "<40-hex merge-base sha>"}}`. The workflow fans out over the seven groups, and each `routing-parity-` job runs its own group twice against the same test harness, once with `litellm/`, `enterprise/`, and `litellm-proxy-extras/` checked out from the base revision and once from the head, with a pytest plugin snapshotting `pg_stat_statements` into `routing-observed.json` per side. The `check` step then compares the two observations and writes `routing-diff.txt`: a statement seen on both sides fails when its role set changed, globally or for the same test (per-test capture is skipped under xdist), unless it is listed in `tests/integration/routing/either_role.json`, where each entry names the statement and a one-line reason it legitimately runs on whichever role asks for it, printed under `== either role ==`. Queries seen on only one side are listed, never failed, `pg_stat_statements` evictions and a role that never ran a statement are failures -### MCP conformance +### MCP regression and conformance baseline + +This gate protects existing behavior while the stateless refactor proceeds. Completion establishes a +passing, enforced baseline; it does not certify exhaustive conformance or future capabilities. +Each later implementation preserves this baseline and adds tests for the behavior it changes The `mcp` group runs a pinned official conformance client against the official reference both directly and through a source-built gateway. Install it with @@ -81,6 +85,44 @@ It preserves protocol versions, arguments, metadata, Origin/Host headers and res No discovery warm-up or expected-failure exemption is used. Missing, failed, skipped or wrongly negotiated required cases fail the MCP integration result. -An installed workflow alone is not a merge gate. After these changes reach the default branch, -verify the hosted MCP job's exact status name and add that check to the existing main ruleset. -Do not mark LIT-7744 complete until all applicable cases pass and that requirement is active. +`mcp/conformance_baseline.json` is the independent required-coverage contract. Its 19 official +scenarios across four upstream revisions, 96 SDK revision/transport combinations, and nine explicit +checks require 181 passing nodes. Test generation does not read this file, so removing a scenario, +transport case or test file cannot silently remove its requirement. Capability mappings name the +required test families that exercise each revision's operations and extensions. A completed +revision or capability without mapped required coverage fails the gate + +To extend the gate, add meaningful behavior assertions to the existing mapped integration file, +then update the baseline with the required cases and corresponding capability mapping in the same +PR. Review that the mapped test actually exercises the capability. A mapping alone is not proof. +Keep required cases passing without skip/xfail exemptions. Track an existing uncovered requirement +with its owning ticket; do not remove baseline protection or advertise unverified new support + +CircleCI currently runs MCP on one node with four pytest workers. The pytest controller combines +all worker reports into `execution.json`; the full invocation checks all required nodes. Missing +worker results, absent files, skips and incomplete runs fail the baseline. Local file selections +check the required cases in those files and are not evidence of a complete gate. If MCP is later +split across CircleCI nodes, aggregate their results against the entire baseline before accepting +the job; per-file successes alone are insufficient + +Remaining coverage grows with the owning implementation: + +| Behavior | Owner and acceptance boundary | +| --- | --- | +| Modern upstream calls without initialization; affected legacy auth/policy parity | LIT-7745, tested with its implementation | +| Authorized list/call across cold replicas without session affinity | LIT-4500, tested with its implementation | +| Full schema/result preservation and remaining upstream pagination | LIT-7750 and LIT-5594 | +| Input relay, MRTR and event/cancellation behavior | LIT-4508, LIT-7752 and LIT-4510 | +| Expanded OAuth and security regressions | LIT-3467, LIT-3559 and LIT-4506 alongside the affected fixes | +| Applicable combined conformance, canary and rollback | LIT-8305 before the corresponding release | + +The SDK matrix provides successful legacy-operation compatibility coverage. It does not repeat +every official rich-content, error, progress or lifecycle assertion for every older client and +transport. Expand those cases when relevant paths change; this limitation does not block unrelated +implementation. Modern revision and extension activation still requires their applicable tests + +An installed workflow alone is not an enforced merge gate. Require the hosted MCP job's exact +status in the existing main ruleset. Close LIT-7744 after this baseline is merged, passes and is +required. LIT-7745 can be developed alongside gate completion; its shared-path changes must pass +the baseline and their added legacy/modern tests before merging. Later coverage extensions belong +to their implementation tickets rather than keeping LIT-7744 open indefinitely diff --git a/tests/integration/_support/conformance.py b/tests/integration/_support/conformance.py index 847129ea333..00d06adabdf 100644 --- a/tests/integration/_support/conformance.py +++ b/tests/integration/_support/conformance.py @@ -4,7 +4,7 @@ import os import queue import signal import subprocess -from collections.abc import AsyncIterator, Iterator +from collections.abc import AsyncIterator, Iterator, Mapping from contextlib import contextmanager from itertools import product from pathlib import Path @@ -12,9 +12,11 @@ from typing import TYPE_CHECKING, Final, Literal import httpx import psutil -from pydantic import BaseModel, Field, JsonValue, TypeAdapter +from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter if TYPE_CHECKING: + from litellm.proxy._experimental.mcp_server.capabilities import RevisionSupport + from integration._support.mcp import McpPeer @@ -219,9 +221,12 @@ class ExecutionReport(BaseModel): def require_passes(directory: Path, required: tuple[str, ...]) -> None: report: Final = ExecutionReport.model_validate_json((directory / "execution.json").read_bytes()) assert required and report.complete, "conformance execution was incomplete" - assert set(required) <= set(report.collected), "conformance cases were not collected" - assert set(required) <= set(report.passed), "conformance cases did not pass" - assert not set(required).intersection(report.skipped), "conformance cases were skipped" + missing: Final = set(required) - set(report.collected) + assert not missing, f"conformance cases were not collected: {sorted(missing)}" + unsuccessful: Final = set(required) - set(report.passed) + assert not unsuccessful, f"conformance cases did not pass: {sorted(unsuccessful)}" + skipped: Final = set(required).intersection(report.skipped) + assert not skipped, f"conformance cases were skipped: {sorted(skipped)}" def translation_cases() -> tuple[tuple[str, str, str, str], ...]: @@ -272,41 +277,74 @@ def official_cases() -> tuple[tuple[str, str], ...]: return tuple(product(OFFICIAL_SCENARIOS, MCP_LEGACY_VERSIONS)) +class CapabilityContract(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + revisions: tuple[str, ...] = Field(min_length=1) + operations: tuple[str, ...] + extensions: tuple[str, ...] + tests: tuple[str, ...] = Field(min_length=1) + + +class ConformanceBaseline(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + revisions: tuple[str, ...] = Field(min_length=1) + ingress_transports: tuple[str, ...] = Field(min_length=1) + upstream_transports: tuple[str, ...] = Field(min_length=1) + official_scenarios: tuple[str, ...] = Field(min_length=1) + capabilities: tuple[CapabilityContract, ...] = Field(min_length=1) + required_tests: tuple[str, ...] = Field(min_length=1) + + +def require_capability_coverage( + baseline: ConformanceBaseline, + required: tuple[str, ...], + support: Mapping[str, "RevisionSupport"], +) -> None: + families: Final = {node.split("[", 1)[0] for node in required} + for contract in baseline.capabilities: + assert set(contract.revisions) <= set(baseline.revisions), "Conformance capability maps an untested revision" + assert set(contract.tests) <= families, "Conformance capability maps a test outside the required baseline" + for revision, advertised in support.items(): + if not advertised.completed: + continue + contracts: Final = tuple(contract for contract in baseline.capabilities if revision in contract.revisions) + assert contracts, f"Conformance capability coverage missing for revision {revision}" + operations: Final = {operation for contract in contracts for operation in contract.operations} + extensions: Final = {extension for contract in contracts for extension in contract.extensions} + assert advertised.operations <= operations, ( + f"Conformance capability coverage missing for {revision}: {sorted(advertised.operations - operations)}" + ) + assert advertised.extensions <= extensions, ( + f"Conformance capability coverage missing for {revision}: {sorted(advertised.extensions - extensions)}" + ) + + def required_conformance_nodes() -> tuple[str, ...]: + from litellm.proxy._experimental.mcp_server.capabilities import REVISION_SUPPORT + + baseline: Final = ConformanceBaseline.model_validate_json( + (Path(__file__).resolve().parents[1] / "mcp/conformance_baseline.json").read_bytes() + ) official: Final = tuple( "tests/integration/mcp/test_mcp_official_conformance.py::test_official_scenario_through_gateway[" + "-".join(case) + "]" - for case in official_cases() + for case in product(baseline.official_scenarios, baseline.revisions) ) matrix: Final = tuple( "tests/integration/mcp/test_mcp_transports.py::test_pinned_revision_pairs_list_and_call_through_gateway[" + "-".join(case) + "]" - for case in translation_cases() - ) - return ( - official - + matrix - + ("tests/integration/mcp/test_mcp_protocol_errors.py::test_omitted_tool_arguments_reach_the_upstream",) - + tuple( - "tests/integration/mcp/test_mcp_protocol_errors.py::test_configured_origin_policy_rejects_before_tool_execution[" - + ingress - + "]" - for ingress in ("server_mcp", "sse") - ) - + tuple( - "tests/integration/mcp/test_mcp_official_conformance.py::" + name - for name in ( - "test_conformance_bridge_preserves_headers_payload_and_error_status[200]", - "test_conformance_bridge_preserves_headers_payload_and_error_status[403]", - "test_official_runner_rejects_unknown_scenario", - "test_official_gateway_session_lifecycle", - "test_stalled_reference_is_killed_and_cannot_report_clean_teardown", - "test_reference_children_are_stopped_after_the_root_exits", - ) + for case in product( + baseline.revisions, baseline.revisions, baseline.upstream_transports, baseline.ingress_transports ) ) + required: Final = official + matrix + baseline.required_tests + assert len(set(required)) == len(required), "Conformance baseline contains duplicate required cases" + require_capability_coverage(baseline, required, REVISION_SUPPORT) + return required def read_negotiation(request: bytes, response: bytes, content_type: str) -> tuple[str, str]: diff --git a/tests/integration/mcp/conformance_baseline.json b/tests/integration/mcp/conformance_baseline.json new file mode 100644 index 00000000000..8a3def5ddcb --- /dev/null +++ b/tests/integration/mcp/conformance_baseline.json @@ -0,0 +1,72 @@ +{ + "revisions": [ + "2024-11-05", + "2025-03-26", + "2025-06-18", + "2025-11-25" + ], + "ingress_transports": [ + "http", + "sse" + ], + "upstream_transports": [ + "http", + "sse", + "stdio" + ], + "official_scenarios": [ + "server-initialize", + "server-sse-multiple-streams", + "ping", + "tools-list", + "tools-call-image", + "tools-call-audio", + "tools-call-embedded-resource", + "tools-call-mixed-content", + "tools-call-error", + "tools-call-with-progress", + "resources-list", + "resources-read-text", + "resources-read-binary", + "resources-templates-read", + "prompts-list", + "prompts-get-simple", + "prompts-get-with-args", + "prompts-get-embedded-resource", + "prompts-get-with-image" + ], + "capabilities": [ + { + "revisions": [ + "2024-11-05", + "2025-03-26", + "2025-06-18", + "2025-11-25" + ], + "operations": [ + "tools/list", + "tools/call", + "prompts/list", + "prompts/get", + "resources/list", + "resources/read", + "resources/templates/list" + ], + "extensions": [], + "tests": [ + "tests/integration/mcp/test_mcp_transports.py::test_pinned_revision_pairs_list_and_call_through_gateway" + ] + } + ], + "required_tests": [ + "tests/integration/mcp/test_mcp_protocol_errors.py::test_omitted_tool_arguments_reach_the_upstream", + "tests/integration/mcp/test_mcp_protocol_errors.py::test_configured_origin_policy_rejects_before_tool_execution[server_mcp]", + "tests/integration/mcp/test_mcp_protocol_errors.py::test_configured_origin_policy_rejects_before_tool_execution[sse]", + "tests/integration/mcp/test_mcp_official_conformance.py::test_conformance_bridge_preserves_headers_payload_and_error_status[200]", + "tests/integration/mcp/test_mcp_official_conformance.py::test_conformance_bridge_preserves_headers_payload_and_error_status[403]", + "tests/integration/mcp/test_mcp_official_conformance.py::test_official_runner_rejects_unknown_scenario", + "tests/integration/mcp/test_mcp_official_conformance.py::test_official_gateway_session_lifecycle", + "tests/integration/mcp/test_mcp_official_conformance.py::test_stalled_reference_is_killed_and_cannot_report_clean_teardown", + "tests/integration/mcp/test_mcp_official_conformance.py::test_reference_children_are_stopped_after_the_root_exits" + ] +} diff --git a/tests/unit/integration_support/test_conformance.py b/tests/unit/integration_support/test_conformance.py index 4ad56699135..791ef5e04cf 100644 --- a/tests/unit/integration_support/test_conformance.py +++ b/tests/unit/integration_support/test_conformance.py @@ -450,3 +450,139 @@ def test_missing_secondary_checks_cannot_report_complete_conformance( read_checks(tmp_path, scenario) report.write_text(json.dumps([{"id": identity, "status": "SUCCESS"} for identity in identities])) assert len(read_checks(tmp_path, scenario)) == len(identities) + + +@pytest.mark.parametrize("removed", ("tools-list", "tools-call-with-progress")) +def test_removing_official_scenario_cannot_shrink_required_baseline( + monkeypatch: pytest.MonkeyPatch, removed: str +) -> None: + from tests.integration._support import conformance + + expected: Final = conformance.required_conformance_nodes() + monkeypatch.setattr( + conformance, "OFFICIAL_SCENARIOS", tuple(s for s in conformance.OFFICIAL_SCENARIOS if s != removed) + ) + assert conformance.required_conformance_nodes() == expected + + +def test_removing_transport_cannot_shrink_required_baseline(monkeypatch: pytest.MonkeyPatch) -> None: + from dataclasses import replace + + from tests.integration._support.conformance import required_conformance_nodes + from litellm.proxy._experimental.mcp_server import capabilities + from litellm.types.mcp import MCPTransport + + expected: Final = required_conformance_nodes() + monkeypatch.setattr( + capabilities, + "REVISION_SUPPORT", + { + revision: replace(support, transports=support.transports - {MCPTransport.stdio}) + for revision, support in capabilities.REVISION_SUPPORT.items() + }, + ) + assert required_conformance_nodes() == expected + + +@pytest.mark.parametrize("kind", ("operations", "extensions")) +def test_new_completed_capability_requires_mapped_tests(monkeypatch: pytest.MonkeyPatch, kind: str) -> None: + from dataclasses import replace + + from tests.integration._support.conformance import required_conformance_nodes + from litellm.proxy._experimental.mcp_server import capabilities + + support: Final = capabilities.REVISION_SUPPORT["2025-11-25"] + expanded: Final = ( + replace(support, operations=support.operations | {"resources/subscribe"}) + if kind == "operations" + else replace(support, extensions=support.extensions | {"tasks"}) + ) + monkeypatch.setattr(capabilities, "REVISION_SUPPORT", {**capabilities.REVISION_SUPPORT, "2025-11-25": expanded}) + with pytest.raises(AssertionError, match="capability"): + required_conformance_nodes() + + +@pytest.mark.parametrize("missing", ("scenario", "file", "worker")) +def test_required_execution_cannot_omit_part_of_baseline(tmp_path: Path, missing: str) -> None: + from tests.integration._support.conformance import required_conformance_nodes + + required: Final = required_conformance_nodes() + omitted: Final = ( + required[:1] + if missing == "scenario" + else tuple(node for node in required if "test_mcp_transports.py" in node) + if missing == "file" + else required[::4] + ) + remaining: Final = tuple(node for node in required if node not in omitted) + (tmp_path / "execution.json").write_text( + json.dumps({"collected": remaining, "passed": remaining, "skipped": [], "complete": True}) + ) + with pytest.raises(AssertionError, match="not collected"): + require_passes(tmp_path, required) + + +@pytest.mark.parametrize("invalid", ("revision", "test")) +def test_capability_mapping_must_reference_required_coverage(invalid: str) -> None: + from tests.integration._support.conformance import ( + CapabilityContract, + ConformanceBaseline, + require_capability_coverage, + ) + + baseline: Final = ConformanceBaseline( + revisions=("2025-11-25",), + ingress_transports=("http",), + upstream_transports=("http",), + official_scenarios=("tools-list",), + required_tests=("test_tools",), + capabilities=( + CapabilityContract( + revisions=("unknown" if invalid == "revision" else "2025-11-25",), + operations=("tools/list",), + extensions=(), + tests=("missing" if invalid == "test" else "test_tools",), + ), + ), + ) + with pytest.raises(AssertionError, match="capability maps"): + require_capability_coverage(baseline, ("test_tools[http]",), {}) + + +def test_capability_can_activate_after_its_required_contract_is_added() -> None: + from dataclasses import replace + + from tests.integration._support.conformance import ( + CapabilityContract, + ConformanceBaseline, + require_capability_coverage, + ) + from litellm.proxy._experimental.mcp_server.capabilities import REVISION_SUPPORT + + baseline: Final = ConformanceBaseline( + revisions=("candidate",), + ingress_transports=("http",), + upstream_transports=("http",), + official_scenarios=("tools-list",), + required_tests=("test_subscription",), + capabilities=( + CapabilityContract( + revisions=("candidate",), + operations=("resources/subscribe",), + extensions=("subscriptions",), + tests=("test_subscription",), + ), + ), + ) + support: Final = replace( + REVISION_SUPPORT["2025-11-25"], + operations=frozenset({"resources/subscribe"}), + extensions=frozenset({"subscriptions"}), + ) + assert require_capability_coverage(baseline, ("test_subscription",), {"candidate": support}) is None + assert ( + require_capability_coverage(baseline, ("test_subscription",), {"future": replace(support, completed=False)}) + is None + ) + with pytest.raises(AssertionError, match="coverage missing for revision"): + require_capability_coverage(baseline, ("test_subscription",), {"future": support})