From 415cf3d6b1c42363e4c8ece5bfdf6b073d0585b9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 25 Apr 2026 04:43:00 +0000 Subject: [PATCH] RALPH: compat matrix slice 2 - add 4 provider columns for basic_messaging_non_streaming (#26478, PRD #26476) Slice 2 of the Claude Code Compatibility Matrix: extend the tracer-bullet cell from slice 1 across all four remaining provider columns for basic_messaging_non_streaming. Proves the multi-provider, multi-model, all-must-pass aggregation logic against a 1x5 grid that exercises every status state. What landed: - tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py - tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py - tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py Per-provider files modeled on test_anthropic.py: each parametrizes over Haiku 4.5 / Sonnet 4.6 / Opus 4.7 (the three Claude tiers required by the PRD), drives the real `claude` CLI through the driver, and reports pass/fail via `compat_result`. Per-cell error strings always include `[]` so the docs tooltip can name the failing model when a cell goes red. - tests/claude_code/basic_messaging_non_streaming/test_azure.py All three (Azure, Claude) cells report `not_applicable` with a reason: Azure OpenAI Service does not host Anthropic models. The test still parametrizes over the same three model ids so the test count per cell is uniform across columns, and a future "Azure adds Anthropic" announcement only requires flipping the body, not the parametrization. - tests/claude_code/sample_compatibility-matrix.json Hand-authored 1x5 sample updated to reflect the slice 2 outcome: anthropic / bedrock_invoke / bedrock_converse / vertex_ai = pass, azure = not_applicable. - tests/claude_code/_builder_unit_tests/test_matrix_builder.py Two new golden-file tests: 1. 1x5 grid: feed the per-model results the four new test files produce on a real run; assert the builder output equals the hand-authored sample byte-for-byte. 2. fail-with-model-named: feed pass/fail/pass for one cell and assert the cell aggregates to fail with the failing model id surfaced in the error string (acceptance criterion: "the error string identifies which model broke"). Key decisions: - Duplication across the four per-provider files is accepted (per the PRD) rather than extracted into a helper. Each file is self-contained so a test author touching one provider doesn't accidentally regress the others. - Per-provider model alias names: `claude--` (e.g. `claude-haiku-4-5-bedrock-invoke`). These are the alias names the proxy operator wires up in the routing config; the test only knows the alias, the proxy knows the upstream model id and region. - Azure is `not_applicable` rather than `not_tested` because the cell will never apply, not "we haven't gotten to it yet" - the two states are visually and semantically distinct in the rendered grid. - Sample shows the realistic best-case outcome (4 pass + 1 NA). The React renderer's coverage of the `fail` and `not_tested` states is exercised by other cells in v1+, not the v0 sample. Tests: 31 -> 34 passing (added 2 builder golden tests + 3 Azure not_applicable parametrizations that pass without env vars). Out of scope per CLAUDE.md (docs live in BerriAI/litellm-docs): - The companion update to compatibility-matrix.json in the docs repo. The hand-authored sample in this repo is the artifact the docs PR copies; opening that doc PR is the next step in this slice. Co-Authored-By: Claude Opus 4.7 --- .../test_matrix_builder.py | 104 ++++++++++++++++++ .../test_azure.py | 44 ++++++++ .../test_bedrock_converse.py | 94 ++++++++++++++++ .../test_bedrock_invoke.py | 94 ++++++++++++++++ .../test_vertex_ai.py | 94 ++++++++++++++++ .../sample_compatibility-matrix.json | 9 +- 6 files changed, 435 insertions(+), 4 deletions(-) create mode 100644 tests/claude_code/basic_messaging_non_streaming/test_azure.py create mode 100644 tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py create mode 100644 tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py create mode 100644 tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py diff --git a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py index c032027e135..87988722b69 100644 --- a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py +++ b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py @@ -175,6 +175,110 @@ def test_load_results_rejects_missing_results_key(tmp_path): load_results(bad) +def test_build_matrix_1x5_grid_matches_published_sample(): + """Slice 2 acceptance: feeding the per-model results that the four + new provider tests produce reproduces the hand-authored 1x5 sample + that the docs page renders. + + Inputs mirror the structure of `compat-results.json` after a real + run with the proxy configured for all five columns: + - anthropic, bedrock_invoke, bedrock_converse, vertex_ai: three + per-model `pass` results each (Haiku, Sonnet, Opus). + - azure: three `not_applicable` results — Azure does not host + Claude, so the column is gray on every row. + + The aggregated matrix must equal the checked-in + `sample_compatibility-matrix.json` byte-for-byte (after JSON load), + so any future schema drift surfaces here in review. + """ + repo_root = Path(__file__).resolve().parents[1] + manifest = load_manifest(repo_root / "manifest.yaml") + + pass_providers = ["anthropic", "bedrock_invoke", "bedrock_converse", "vertex_ai"] + models = ["claude-haiku-4-5", "claude-sonnet-4-6", "claude-opus-4-7"] + azure_reason = ( + "Azure OpenAI Service does not host Anthropic Claude models. " + "Route Claude requests through Anthropic, AWS Bedrock, or GCP Vertex AI." + ) + + results = [] + for provider in pass_providers: + for model in models: + results.append( + { + "feature_id": "basic_messaging_non_streaming", + "provider": provider, + "nodeid": ( + f"tests/claude_code/basic_messaging_non_streaming/test_{provider}.py" + f"::test[{model}]" + ), + "result": {"status": "pass"}, + } + ) + for model in models: + results.append( + { + "feature_id": "basic_messaging_non_streaming", + "provider": "azure", + "nodeid": ( + "tests/claude_code/basic_messaging_non_streaming/test_azure.py" + f"::test[{model}]" + ), + "result": {"status": "not_applicable", "reason": azure_reason}, + } + ) + + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v1.83.0-stable", + claude_code_version="2.1.120", + generated_at="2026-04-25T00:00:00Z", + ) + expected = json.loads((repo_root / "sample_compatibility-matrix.json").read_text()) + assert matrix == expected + + +def test_build_matrix_1x5_grid_one_failing_model_breaks_cell(): + """If even one of three models fails on a provider, that cell is fail + and the error string carries the failing model id so the docs + tooltip can name the outlier.""" + repo_root = Path(__file__).resolve().parents[1] + manifest = load_manifest(repo_root / "manifest.yaml") + + results = [ + { + "feature_id": "basic_messaging_non_streaming", + "provider": "bedrock_invoke", + "result": {"status": "pass"}, + }, + { + "feature_id": "basic_messaging_non_streaming", + "provider": "bedrock_invoke", + "result": { + "status": "fail", + "error": "[claude-opus-4-7-bedrock-invoke] claude CLI exited 1: throttled", + }, + }, + { + "feature_id": "basic_messaging_non_streaming", + "provider": "bedrock_invoke", + "result": {"status": "pass"}, + }, + ] + + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + cell = matrix["features"][0]["providers"]["bedrock_invoke"] + assert cell["status"] == "fail" + assert "claude-opus-4-7-bedrock-invoke" in cell["error"] + + def test_build_from_paths_writes_output(tmp_path): out = tmp_path / "compatibility-matrix.json" matrix = build_from_paths( diff --git a/tests/claude_code/basic_messaging_non_streaming/test_azure.py b/tests/claude_code/basic_messaging_non_streaming/test_azure.py new file mode 100644 index 00000000000..5b3bd509bac --- /dev/null +++ b/tests/claude_code/basic_messaging_non_streaming/test_azure.py @@ -0,0 +1,44 @@ +"""basic_messaging_non_streaming x Azure. + +Azure (Azure OpenAI Service) does not host Anthropic Claude models — +the platform's first-party catalog is OpenAI models, plus a smaller set +of Microsoft and partner models. There is no supported route for the +`claude` CLI to talk to Claude through Azure via LiteLLM, so every +(model, Azure) combination for `basic_messaging_non_streaming` reports +`not_applicable` rather than `fail`. This is what the matrix's +`not_applicable` state is for: the renderer paints the cell gray with a +tooltip explaining the cell will never apply, distinct from a +genuine regression. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/basic_messaging_non_streaming/test_azure.py + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^ + feature_id provider +""" + +from __future__ import annotations + +import pytest + +# Same three Claude tiers the other provider columns parametrize over, +# so the test count per cell stays uniform across the matrix and any +# future "Azure adds Anthropic" announcement only requires flipping the +# body, not the parametrization. +AZURE_MODELS = [ + "claude-haiku-4-5", + "claude-sonnet-4-6", + "claude-opus-4-7", +] + +NOT_APPLICABLE_REASON = ( + "Azure OpenAI Service does not host Anthropic Claude models. " + "Route Claude requests through Anthropic, AWS Bedrock, or GCP Vertex AI." +) + + +@pytest.mark.parametrize("model", AZURE_MODELS) +def test_basic_messaging_non_streaming_azure(compat_result, model): + """Report `not_applicable` for every (model, Azure) combination.""" + compat_result.set({"status": "not_applicable", "reason": NOT_APPLICABLE_REASON}) diff --git a/tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py b/tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py new file mode 100644 index 00000000000..dc604f65acb --- /dev/null +++ b/tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py @@ -0,0 +1,94 @@ +"""basic_messaging_non_streaming x Bedrock (Converse). + +Drive the real `claude` CLI in headless mode against a running LiteLLM +proxy that routes Claude requests to AWS Bedrock via the unified +`Converse` API path, and report the outcome via `compat_result`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider + +Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus +4.7; the cell only goes green if all three pass. We parametrize over the +three models and the conftest aggregator produces one cell from the +three results, naming the failing model in the error string when any +model fails. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.cli_driver import ClaudeCLIError, run_claude + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Per-model aliases registered in the LiteLLM proxy's routing config to +# point at Bedrock's Converse endpoint. The driver only sends the alias; +# the proxy is the one that knows the upstream model id and routing +# strategy. +BEDROCK_CONVERSE_MODELS = [ + "claude-haiku-4-5-bedrock-converse", + "claude-sonnet-4-6-bedrock-converse", + "claude-opus-4-7-bedrock-converse", +] + + +@pytest.mark.parametrize("model", BEDROCK_CONVERSE_MODELS) +def test_basic_messaging_non_streaming_bedrock_converse(compat_result, model): + """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False + ) + + try: + result = run_claude( + prompt="Reply with the single word 'pong' and nothing else.", + model=model, + base_url=base_url, + api_key=api_key, + ) + except ClaudeCLIError as exc: + compat_result.set({"status": "fail", "error": f"[{model}] {exc}"}) + pytest.fail(str(exc), pytrace=False) + return + + if result.exit_code != 0: + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}", + } + ) + pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False) + return + + if not result.text.strip(): + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude returned empty assistant text", + } + ) + pytest.fail(f"empty reply for {model}", pytrace=False) + return + + compat_result.set({"status": "pass"}) diff --git a/tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py b/tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py new file mode 100644 index 00000000000..8dc4ef9fff0 --- /dev/null +++ b/tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py @@ -0,0 +1,94 @@ +"""basic_messaging_non_streaming x Bedrock (Invoke). + +Drive the real `claude` CLI in headless mode against a running LiteLLM +proxy that routes Claude requests to AWS Bedrock via the legacy +`InvokeModel` API path, and report the outcome via `compat_result`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider + +Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus +4.7; the cell only goes green if all three pass. We parametrize over the +three models and the conftest aggregator produces one cell from the +three results, naming the failing model in the error string when any +model fails. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.cli_driver import ClaudeCLIError, run_claude + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Per-model aliases registered in the LiteLLM proxy's routing config to +# point at Bedrock's legacy InvokeModel endpoint. The driver only sends +# the alias; the proxy is the one that knows the upstream model id and +# routing strategy. +BEDROCK_INVOKE_MODELS = [ + "claude-haiku-4-5-bedrock-invoke", + "claude-sonnet-4-6-bedrock-invoke", + "claude-opus-4-7-bedrock-invoke", +] + + +@pytest.mark.parametrize("model", BEDROCK_INVOKE_MODELS) +def test_basic_messaging_non_streaming_bedrock_invoke(compat_result, model): + """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False + ) + + try: + result = run_claude( + prompt="Reply with the single word 'pong' and nothing else.", + model=model, + base_url=base_url, + api_key=api_key, + ) + except ClaudeCLIError as exc: + compat_result.set({"status": "fail", "error": f"[{model}] {exc}"}) + pytest.fail(str(exc), pytrace=False) + return + + if result.exit_code != 0: + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}", + } + ) + pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False) + return + + if not result.text.strip(): + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude returned empty assistant text", + } + ) + pytest.fail(f"empty reply for {model}", pytrace=False) + return + + compat_result.set({"status": "pass"}) diff --git a/tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py b/tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py new file mode 100644 index 00000000000..a32d8d720ff --- /dev/null +++ b/tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py @@ -0,0 +1,94 @@ +"""basic_messaging_non_streaming x Vertex AI. + +Drive the real `claude` CLI in headless mode against a running LiteLLM +proxy that routes Claude requests to Anthropic's models on Google Cloud +Vertex AI, and report the outcome via `compat_result`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/basic_messaging_non_streaming/test_vertex_ai.py + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus +4.7; the cell only goes green if all three pass. We parametrize over the +three models and the conftest aggregator produces one cell from the +three results, naming the failing model in the error string when any +model fails. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.cli_driver import ClaudeCLIError, run_claude + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Per-model aliases registered in the LiteLLM proxy's routing config to +# point at Vertex AI's Anthropic model endpoints. The driver only sends +# the alias; the proxy is the one that knows the upstream publisher +# model id and the GCP region. +VERTEX_AI_MODELS = [ + "claude-haiku-4-5-vertex", + "claude-sonnet-4-6-vertex", + "claude-opus-4-7-vertex", +] + + +@pytest.mark.parametrize("model", VERTEX_AI_MODELS) +def test_basic_messaging_non_streaming_vertex_ai(compat_result, model): + """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False + ) + + try: + result = run_claude( + prompt="Reply with the single word 'pong' and nothing else.", + model=model, + base_url=base_url, + api_key=api_key, + ) + except ClaudeCLIError as exc: + compat_result.set({"status": "fail", "error": f"[{model}] {exc}"}) + pytest.fail(str(exc), pytrace=False) + return + + if result.exit_code != 0: + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}", + } + ) + pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False) + return + + if not result.text.strip(): + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude returned empty assistant text", + } + ) + pytest.fail(f"empty reply for {model}", pytrace=False) + return + + compat_result.set({"status": "pass"}) diff --git a/tests/claude_code/sample_compatibility-matrix.json b/tests/claude_code/sample_compatibility-matrix.json index d4e38818e8e..7503bb3c42a 100644 --- a/tests/claude_code/sample_compatibility-matrix.json +++ b/tests/claude_code/sample_compatibility-matrix.json @@ -19,16 +19,17 @@ "status": "pass" }, "bedrock_invoke": { - "status": "not_tested" + "status": "pass" }, "bedrock_converse": { - "status": "not_tested" + "status": "pass" }, "vertex_ai": { - "status": "not_tested" + "status": "pass" }, "azure": { - "status": "not_tested" + "status": "not_applicable", + "reason": "Azure OpenAI Service does not host Anthropic Claude models. Route Claude requests through Anthropic, AWS Bedrock, or GCP Vertex AI." } } }