diff --git a/.gitignore b/.gitignore index 38bf9554b5b..5f07ee4007c 100644 --- a/.gitignore +++ b/.gitignore @@ -100,4 +100,7 @@ STABILIZATION_TODO.md **/test-results **/playwright-report **/*.storageState.json -**/coverage \ No newline at end of file +**/coverage + +# Claude Code compatibility-matrix pytest artifact (CI-only output). +compat-results.json \ No newline at end of file diff --git a/tests/claude_code/__init__.py b/tests/claude_code/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/_builder_unit_tests/__init__.py b/tests/claude_code/_builder_unit_tests/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json b/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json new file mode 100644 index 00000000000..d3aca0142dc --- /dev/null +++ b/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json @@ -0,0 +1,38 @@ +{ + "schema_version": "1", + "generated_at": "2026-04-25T00:00:00Z", + "litellm_version": "v1.83.0-stable", + "claude_code_version": "2.1.120", + "providers": [ + "anthropic", + "bedrock_invoke" + ], + "features": [ + { + "id": "basic_messaging_non_streaming", + "name": "Basic messaging (non-streaming)", + "providers": { + "anthropic": { + "status": "pass" + }, + "bedrock_invoke": { + "status": "not_tested" + } + } + }, + { + "id": "tool_use", + "name": "Tool use", + "providers": { + "anthropic": { + "status": "fail", + "error": "[claude-sonnet-4-6] tool call dropped" + }, + "bedrock_invoke": { + "status": "not_applicable", + "reason": "tool use not yet wired up for Bedrock Invoke" + } + } + } + ] +} diff --git a/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml b/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml new file mode 100644 index 00000000000..e88bdc6ddf5 --- /dev/null +++ b/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml @@ -0,0 +1,9 @@ +schema_version: "1" +providers: + - anthropic + - bedrock_invoke +features: + - id: basic_messaging_non_streaming + name: Basic messaging (non-streaming) + - id: tool_use + name: Tool use diff --git a/tests/claude_code/_builder_unit_tests/fixtures/results.json b/tests/claude_code/_builder_unit_tests/fixtures/results.json new file mode 100644 index 00000000000..2ede64f167c --- /dev/null +++ b/tests/claude_code/_builder_unit_tests/fixtures/results.json @@ -0,0 +1,41 @@ +{ + "schema_version": "1", + "results": [ + { + "feature_id": "basic_messaging_non_streaming", + "provider": "anthropic", + "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-haiku-4-5]", + "result": {"status": "pass"} + }, + { + "feature_id": "basic_messaging_non_streaming", + "provider": "anthropic", + "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-sonnet-4-6]", + "result": {"status": "pass"} + }, + { + "feature_id": "basic_messaging_non_streaming", + "provider": "anthropic", + "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-opus-4-7]", + "result": {"status": "pass"} + }, + { + "feature_id": "tool_use", + "provider": "anthropic", + "nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-haiku-4-5]", + "result": {"status": "pass"} + }, + { + "feature_id": "tool_use", + "provider": "anthropic", + "nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-sonnet-4-6]", + "result": {"status": "fail", "error": "[claude-sonnet-4-6] tool call dropped"} + }, + { + "feature_id": "tool_use", + "provider": "bedrock_invoke", + "nodeid": "tests/claude_code/tool_use/test_bedrock_invoke.py::test_x[claude-haiku-4-5]", + "result": {"status": "not_applicable", "reason": "tool use not yet wired up for Bedrock Invoke"} + } + ] +} diff --git a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py new file mode 100644 index 00000000000..c032027e135 --- /dev/null +++ b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py @@ -0,0 +1,192 @@ +"""Golden-file tests for the Matrix JSON Builder. + +These tests fix the published JSON schema. The builder is a pure function +from (manifest, results, metadata) → matrix dict, so we feed it a fixture +input set and compare the produced dict to a checked-in expected output. + +Any schema drift — intentional or accidental — surfaces as a diff in PR +review. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from tests.claude_code.matrix_builder import ( + ManifestError, + ResultsError, + build_from_paths, + build_matrix, + load_manifest, + load_results, +) + +FIXTURES = Path(__file__).parent / "fixtures" + + +def test_build_matrix_matches_golden_file(tmp_path): + manifest = load_manifest(FIXTURES / "manifest.yaml") + results = load_results(FIXTURES / "results.json") + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v1.83.0-stable", + claude_code_version="2.1.120", + generated_at="2026-04-25T00:00:00Z", + ) + expected = json.loads((FIXTURES / "expected_matrix.json").read_text()) + assert matrix == expected + + +def test_build_matrix_pass_requires_all_models_pass(): + """Multiple results in one cell must all be pass for the cell to be pass.""" + manifest = { + "schema_version": "1", + "providers": ["anthropic"], + "features": [{"id": "f", "name": "F"}], + } + results = [ + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + ] + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"} + + +def test_build_matrix_any_fail_makes_cell_fail(): + manifest = { + "schema_version": "1", + "providers": ["anthropic"], + "features": [{"id": "f", "name": "F"}], + } + results = [ + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + { + "feature_id": "f", + "provider": "anthropic", + "result": {"status": "fail", "error": "[claude-opus-4-7] timeout"}, + }, + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + ] + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + cell = matrix["features"][0]["providers"]["anthropic"] + assert cell["status"] == "fail" + assert cell["error"] == "[claude-opus-4-7] timeout" + + +def test_build_matrix_fills_not_tested_for_missing_cells(): + manifest = { + "schema_version": "1", + "providers": ["anthropic", "azure"], + "features": [{"id": "f", "name": "F"}], + } + results = [ + {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}}, + ] + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + cells = matrix["features"][0]["providers"] + assert cells["anthropic"] == {"status": "pass"} + assert cells["azure"] == {"status": "not_tested"} + + +def test_build_matrix_preserves_provider_and_feature_order(): + manifest = { + "schema_version": "1", + "providers": ["azure", "anthropic", "vertex_ai"], + "features": [ + {"id": "z", "name": "Z"}, + {"id": "a", "name": "A"}, + ], + } + matrix = build_matrix( + manifest=manifest, + results=[], + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + assert matrix["providers"] == ["azure", "anthropic", "vertex_ai"] + assert [f["id"] for f in matrix["features"]] == ["z", "a"] + assert list(matrix["features"][0]["providers"].keys()) == [ + "azure", + "anthropic", + "vertex_ai", + ] + + +def test_build_matrix_emits_schema_version_one(): + manifest = { + "schema_version": "1", + "providers": ["anthropic"], + "features": [{"id": "f", "name": "F"}], + } + matrix = build_matrix( + manifest=manifest, + results=[], + litellm_version="v", + claude_code_version="c", + generated_at="t", + ) + assert matrix["schema_version"] == "1" + + +def test_load_manifest_rejects_wrong_schema_version(tmp_path): + bad = tmp_path / "manifest.yaml" + bad.write_text( + 'schema_version: "2"\nproviders: [anthropic]\nfeatures:\n - id: f\n name: F\n' + ) + with pytest.raises(ManifestError, match="schema_version"): + load_manifest(bad) + + +def test_load_manifest_rejects_empty_features(tmp_path): + bad = tmp_path / "manifest.yaml" + bad.write_text('schema_version: "1"\nproviders: [anthropic]\nfeatures: []\n') + with pytest.raises(ManifestError): + load_manifest(bad) + + +def test_load_results_rejects_missing_results_key(tmp_path): + bad = tmp_path / "results.json" + bad.write_text(json.dumps({"schema_version": "1"})) + with pytest.raises(ResultsError): + load_results(bad) + + +def test_build_from_paths_writes_output(tmp_path): + out = tmp_path / "compatibility-matrix.json" + matrix = build_from_paths( + manifest_path=FIXTURES / "manifest.yaml", + results_path=FIXTURES / "results.json", + litellm_version="v1.83.0-stable", + claude_code_version="2.1.120", + generated_at="2026-04-25T00:00:00Z", + output_path=out, + ) + assert out.exists() + on_disk = json.loads(out.read_text()) + assert on_disk == matrix + expected = json.loads((FIXTURES / "expected_matrix.json").read_text()) + assert on_disk == expected diff --git a/tests/claude_code/_driver_unit_tests/__init__.py b/tests/claude_code/_driver_unit_tests/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/_driver_unit_tests/test_cli_driver.py b/tests/claude_code/_driver_unit_tests/test_cli_driver.py new file mode 100644 index 00000000000..fbdea1a0aa4 --- /dev/null +++ b/tests/claude_code/_driver_unit_tests/test_cli_driver.py @@ -0,0 +1,239 @@ +"""Unit tests for the Claude Code CLI Driver. + +These tests mock the subprocess so they run anywhere — no network, no +`claude` install, no API keys. They cover the behavior contract: +argument assembly, environment overlay, stream-JSON parsing, exit-code +plumbing, and the structured failure modes (CLI not found, timeout). +""" + +from __future__ import annotations + +import json +import subprocess +from dataclasses import dataclass +from typing import List, Optional + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + DriverResult, + run_claude, +) + + +@dataclass +class _Completed: + returncode: int = 0 + stdout: str = "" + stderr: str = "" + + +def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""): + captured = {} + + def runner(cmd, env, capture_output, text, timeout, check): + captured["cmd"] = cmd + captured["env"] = env + captured["timeout"] = timeout + return _Completed(returncode=returncode, stdout=stdout, stderr=stderr) + + return runner, captured + + +def test_run_claude_assembles_command_correctly(): + runner, captured = _make_runner( + stdout='{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}\n' + ) + run_claude( + prompt="hello", + model="claude-haiku-4-5", + base_url="http://localhost:4000", + api_key="sk-test", + runner=runner, + ) + cmd = captured["cmd"] + assert cmd[0] == "claude" + assert "--print" in cmd + assert "--output-format" in cmd + assert "stream-json" in cmd + assert "--model" in cmd + assert "claude-haiku-4-5" in cmd + assert cmd[-1] == "hello" + + +def test_run_claude_overlays_proxy_env(): + runner, captured = _make_runner(stdout="") + run_claude( + prompt="hi", + model="claude-opus-4-7", + base_url="http://proxy.example:4000", + api_key="sk-abc", + runner=runner, + ) + env = captured["env"] + assert env["ANTHROPIC_BASE_URL"] == "http://proxy.example:4000" + assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-abc" + + +def test_run_claude_extra_env_takes_precedence_over_os_environ(monkeypatch): + monkeypatch.setenv("FOO", "from-os") + runner, captured = _make_runner(stdout="") + run_claude( + prompt="hi", + model="claude-opus-4-7", + base_url="http://localhost", + api_key="sk-abc", + extra_env={"FOO": "from-arg"}, + runner=runner, + ) + assert captured["env"]["FOO"] == "from-arg" + + +def test_run_claude_parses_stream_json_assistant_text(): + events = [ + {"type": "system", "session_id": "abc"}, + { + "type": "assistant", + "message": { + "content": [ + {"type": "text", "text": "Hello "}, + {"type": "text", "text": "world"}, + ] + }, + }, + {"type": "result", "usage": {"input_tokens": 10, "output_tokens": 2}}, + ] + stdout = "\n".join(json.dumps(e) for e in events) + "\n" + runner, _ = _make_runner(stdout=stdout) + result = run_claude( + prompt="hi", + model="claude-haiku-4-5", + base_url="http://localhost", + api_key="sk-abc", + runner=runner, + ) + assert isinstance(result, DriverResult) + assert result.text == "Hello world" + assert len(result.events) == 3 + assert result.usage == {"input_tokens": 10, "output_tokens": 2} + assert result.exit_code == 0 + + +def test_run_claude_handles_string_message_content(): + """Some CLI versions emit `message.content` as a plain string.""" + stdout = ( + json.dumps({"type": "assistant", "message": {"content": "bare text"}}) + "\n" + ) + runner, _ = _make_runner(stdout=stdout) + result = run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="k", + runner=runner, + ) + assert result.text == "bare text" + + +def test_run_claude_skips_malformed_lines(): + stdout = ( + "not-json\n" + + json.dumps( + { + "type": "assistant", + "message": {"content": [{"type": "text", "text": "x"}]}, + } + ) + + "\n" + + "{also-bad\n" + ) + runner, _ = _make_runner(stdout=stdout) + result = run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="k", + runner=runner, + ) + assert result.text == "x" + assert len(result.events) == 1 + + +def test_run_claude_propagates_nonzero_exit_code(): + runner, _ = _make_runner(stdout="", returncode=2, stderr="auth failed") + result = run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="k", + runner=runner, + ) + assert result.exit_code == 2 + assert result.stderr == "auth failed" + assert result.text == "" + + +def test_run_claude_raises_on_missing_cli(): + def runner(*args, **kwargs): + raise FileNotFoundError(2, "no such file", "claude") + + with pytest.raises(ClaudeCLIError, match="claude CLI not found"): + run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="k", + runner=runner, + ) + + +def test_run_claude_raises_on_timeout(): + def runner(*args, **kwargs): + raise subprocess.TimeoutExpired(cmd="claude", timeout=1) + + with pytest.raises(ClaudeCLIError, match="timed out"): + run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="k", + timeout=1, + runner=runner, + ) + + +def test_run_claude_validates_required_params(): + runner, _ = _make_runner() + with pytest.raises(ValueError, match="prompt"): + run_claude( + prompt="", + model="m", + base_url="http://x", + api_key="k", + runner=runner, + ) + with pytest.raises(ValueError, match="model"): + run_claude( + prompt="hi", + model="", + base_url="http://x", + api_key="k", + runner=runner, + ) + with pytest.raises(ValueError, match="base_url"): + run_claude( + prompt="hi", + model="m", + base_url="", + api_key="k", + runner=runner, + ) + with pytest.raises(ValueError, match="api_key"): + run_claude( + prompt="hi", + model="m", + base_url="http://x", + api_key="", + runner=runner, + ) diff --git a/tests/claude_code/_driver_unit_tests/test_compat_result.py b/tests/claude_code/_driver_unit_tests/test_compat_result.py new file mode 100644 index 00000000000..04e19bdc482 --- /dev/null +++ b/tests/claude_code/_driver_unit_tests/test_compat_result.py @@ -0,0 +1,70 @@ +"""Tests for the `compat_result` fixture's tagged-union validation. + +The conftest's `pytest_runtest_makereport` hook is exercised end-to-end by +the matrix-builder golden-file tests (which consume a results.json that +the harness would produce). Here we just test the input-validation +contract on `CompatResult.set()`. +""" + +from __future__ import annotations + +import pytest + +from tests.claude_code.conftest import CompatResult + + +def test_set_pass_is_accepted(): + r = CompatResult() + r.set({"status": "pass"}) + assert r.value == {"status": "pass"} + + +def test_set_fail_requires_error(): + r = CompatResult() + with pytest.raises(ValueError, match="requires 'error'"): + r.set({"status": "fail"}) + + +def test_set_fail_with_error_is_accepted(): + r = CompatResult() + r.set({"status": "fail", "error": "boom"}) + assert r.value == {"status": "fail", "error": "boom"} + + +def test_set_not_applicable_requires_reason(): + r = CompatResult() + with pytest.raises(ValueError, match="requires 'reason'"): + r.set({"status": "not_applicable"}) + + +def test_set_not_applicable_with_reason_is_accepted(): + r = CompatResult() + r.set({"status": "not_applicable", "reason": "Bedrock has no /thinking"}) + assert r.value == {"status": "not_applicable", "reason": "Bedrock has no /thinking"} + + +def test_set_not_tested_is_accepted(): + r = CompatResult() + r.set({"status": "not_tested"}) + assert r.value == {"status": "not_tested"} + + +def test_set_rejects_unknown_status(): + r = CompatResult() + with pytest.raises(ValueError, match="status must be one of"): + r.set({"status": "maybe"}) + + +def test_set_rejects_non_dict(): + r = CompatResult() + with pytest.raises(TypeError): + r.set("pass") # type: ignore[arg-type] + + +def test_set_copies_input(): + """Mutating the dict after set() must not change the stored value.""" + r = CompatResult() + payload = {"status": "fail", "error": "x"} + r.set(payload) + payload["error"] = "mutated" + assert r.value["error"] == "x" diff --git a/tests/claude_code/basic_messaging_non_streaming/__init__.py b/tests/claude_code/basic_messaging_non_streaming/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py b/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py new file mode 100644 index 00000000000..7e0a1f16583 --- /dev/null +++ b/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py @@ -0,0 +1,98 @@ +"""basic_messaging_non_streaming × Anthropic. + +The thinnest end-to-end path through every layer of the matrix: drive the +real `claude` CLI in headless mode against a running LiteLLM proxy that +routes to Anthropic, and report the outcome via `compat_result`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/basic_messaging_non_streaming/test_anthropic.py + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus +4.7; the cell only goes green if all three pass. We parametrize over the +three models and the conftest aggregator produces one cell from the three +results. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.cli_driver import ClaudeCLIError, run_claude + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Per the PRD: each cell is exercised against three Claude tiers via the +# Anthropic provider. Aliases are configured in the LiteLLM proxy's +# routing config; the driver only sends the alias. +ANTHROPIC_MODELS = [ + "claude-haiku-4-5", + "claude-sonnet-4-6", + "claude-opus-4-7", +] + + +@pytest.mark.parametrize("model", ANTHROPIC_MODELS) +def test_basic_messaging_non_streaming_anthropic(compat_result, model): + """Drive the `claude` CLI against the LiteLLM proxy and assert a reply. + + "Basic messaging" means: send a single user prompt, receive any + non-empty assistant text reply, no tools, no streaming, no thinking. + The whole point of this slice is to prove the path works at all — + so the assertion is intentionally lenient on the reply contents. + """ + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False + ) + + try: + result = run_claude( + prompt="Reply with the single word 'pong' and nothing else.", + model=model, + base_url=base_url, + api_key=api_key, + ) + except ClaudeCLIError as exc: + compat_result.set({"status": "fail", "error": f"[{model}] {exc}"}) + pytest.fail(str(exc), pytrace=False) + return + + if result.exit_code != 0: + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}", + } + ) + pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False) + return + + if not result.text.strip(): + compat_result.set( + { + "status": "fail", + "error": f"[{model}] claude returned empty assistant text", + } + ) + pytest.fail(f"empty reply for {model}", pytrace=False) + return + + compat_result.set({"status": "pass"}) diff --git a/tests/claude_code/cli_driver.py b/tests/claude_code/cli_driver.py new file mode 100644 index 00000000000..cd8732491d3 --- /dev/null +++ b/tests/claude_code/cli_driver.py @@ -0,0 +1,194 @@ +"""Claude Code CLI Driver. + +A thin wrapper around the `claude` CLI in headless mode. Every compatibility +test consumes only this module — tests must never shell out directly. This +keeps the subprocess assembly, stream-JSON parsing, and result shape in a +single place that can be unit-tested with a mocked subprocess. + +The driver is deliberately small: it knows how to invoke the CLI, drain its +stream-JSON output, and return a structured `DriverResult`. Higher-level +matrix concerns (status aggregation, manifest lookup, JSON serialization) +live in `matrix_builder.py`. +""" + +from __future__ import annotations + +import json +import os +import subprocess +from dataclasses import dataclass, field +from typing import Any, Dict, List, Mapping, Optional, Sequence + +CLAUDE_CLI_DEFAULT = "claude" +DEFAULT_TIMEOUT_SECONDS = 120 + + +class ClaudeCLIError(RuntimeError): + """Raised when the `claude` CLI cannot be invoked or returns a fatal error.""" + + +@dataclass +class DriverResult: + """Structured outcome of a single `claude` CLI invocation. + + `text` is the assistant's final user-visible reply (joined across any + intermediate `assistant` events for non-streaming runs). `events` is the + raw list of stream-JSON objects emitted by the CLI, preserved so test + authors can write feature-specific assertions (tool calls, cache hits, + usage) without re-parsing stdout. + """ + + text: str + events: List[Dict[str, Any]] = field(default_factory=list) + exit_code: int = 0 + stderr: str = "" + usage: Optional[Dict[str, Any]] = None + duration_ms: Optional[int] = None + + +def run_claude( + *, + prompt: str, + model: str, + base_url: str, + api_key: str, + extra_env: Optional[Mapping[str, str]] = None, + extra_args: Optional[Sequence[str]] = None, + cli_path: str = CLAUDE_CLI_DEFAULT, + timeout: float = DEFAULT_TIMEOUT_SECONDS, + runner: Optional[Any] = None, +) -> DriverResult: + """Invoke `claude` once in headless stream-JSON mode and return the result. + + The CLI is pointed at a LiteLLM proxy via `ANTHROPIC_BASE_URL` / + `ANTHROPIC_AUTH_TOKEN`, so the same code path exercises every provider + column — only the model id and the proxy's routing differ between + invocations. + + `runner` is an injection seam used by the unit tests: by default we call + `subprocess.run`, but the test suite swaps in a fake that yields canned + stream-JSON. Production callers should never set it. + """ + if not prompt: + raise ValueError("prompt must be a non-empty string") + if not model: + raise ValueError("model must be a non-empty string") + if not base_url: + raise ValueError("base_url must be a non-empty string") + if not api_key: + raise ValueError("api_key must be a non-empty string") + + cmd: List[str] = [ + cli_path, + "--print", + "--output-format", + "stream-json", + "--verbose", + "--model", + model, + prompt, + ] + if extra_args: + cmd.extend(extra_args) + + env = {**os.environ, **(extra_env or {})} + env["ANTHROPIC_BASE_URL"] = base_url + env["ANTHROPIC_AUTH_TOKEN"] = api_key + + run_fn = runner or subprocess.run + try: + completed = run_fn( + cmd, + env=env, + capture_output=True, + text=True, + timeout=timeout, + check=False, + ) + except FileNotFoundError as exc: + raise ClaudeCLIError( + f"claude CLI not found at {cli_path!r}; install with `npm i -g @anthropic-ai/claude-code`" + ) from exc + except subprocess.TimeoutExpired as exc: + raise ClaudeCLIError(f"claude CLI timed out after {timeout}s") from exc + + events = _parse_stream_json(completed.stdout or "") + text = _extract_assistant_text(events) + usage = _extract_usage(events) + + return DriverResult( + text=text, + events=events, + exit_code=completed.returncode, + stderr=completed.stderr or "", + usage=usage, + ) + + +def _parse_stream_json(stdout: str) -> List[Dict[str, Any]]: + """Parse newline-delimited JSON emitted by `claude --output-format stream-json`. + + Lines that don't parse as JSON are silently skipped — the CLI occasionally + emits debug output we don't care about, and a single malformed line should + not abort the whole run. Real failure modes surface via exit code. + """ + events: List[Dict[str, Any]] = [] + for line in stdout.splitlines(): + line = line.strip() + if not line: + continue + try: + obj = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(obj, dict): + events.append(obj) + return events + + +def _extract_assistant_text(events: Sequence[Mapping[str, Any]]) -> str: + """Concatenate the text content of every `assistant` event in order. + + The non-streaming `--print` path emits a single `assistant` event whose + `message.content` is a list of content blocks. We walk the blocks and + join every `text` block — the CLI prints other block types (e.g. + `tool_use`) which we ignore for the basic-messaging case. + """ + chunks: List[str] = [] + for event in events: + if event.get("type") != "assistant": + continue + message = event.get("message") or {} + content = message.get("content") + if isinstance(content, str): + chunks.append(content) + continue + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict): + continue + if block.get("type") == "text" and isinstance(block.get("text"), str): + chunks.append(block["text"]) + return "".join(chunks) + + +def _extract_usage(events: Sequence[Mapping[str, Any]]) -> Optional[Dict[str, Any]]: + """Return the most recent `usage` block seen on any event, if any. + + The CLI surfaces token + cache usage on the final `result` event for + non-streaming runs, but earlier events also carry partial usage in some + versions; taking the last non-empty one is the safe default. + """ + last: Optional[Dict[str, Any]] = None + for event in events: + usage = event.get("usage") + if isinstance(usage, dict) and usage: + last = usage + continue + message = event.get("message") + if isinstance(message, dict): + inner = message.get("usage") + if isinstance(inner, dict) and inner: + last = inner + return last diff --git a/tests/claude_code/conftest.py b/tests/claude_code/conftest.py new file mode 100644 index 00000000000..567c24720ed --- /dev/null +++ b/tests/claude_code/conftest.py @@ -0,0 +1,164 @@ +"""Pytest plumbing for the Claude Code compatibility matrix. + +Two responsibilities live here: + +1. The `compat_result` fixture — the only API a test author needs to learn. + Tests call `compat_result.set({"status": "pass"})` (or fail / not_applicable) + to report their outcome as a tagged union. The fixture is per-test and + stores the last value reported. + +2. The `pytest_runtest_makereport` hook — captures each test's reported result, + infers (feature, provider) from the file path, and writes a single + `compat-results.json` artifact next to JUnit XML. The Matrix JSON Builder + consumes this artifact to produce the published `compatibility-matrix.json`. + +The (feature, provider) inference comes from the test file path: the parent +directory name is the feature_id (matching `manifest.yaml`), and the file +stem after the leading `test_` is the provider id. This avoids per-file +metadata that drifts. +""" + +from __future__ import annotations + +import json +import os +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, List, Optional + +import pytest + +VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"} +RESULTS_ARTIFACT_ENV = "COMPAT_RESULTS_PATH" +DEFAULT_ARTIFACT_PATH = "compat-results.json" + + +@dataclass +class CompatResult: + """Per-test recorder for compatibility outcomes. + + Tests interact only via `.set(...)`. `.value` is read by the + `pytest_runtest_makereport` hook after the test body finishes. + """ + + value: Optional[Dict[str, Any]] = None + + def set(self, result: Dict[str, Any]) -> None: + if not isinstance(result, dict): + raise TypeError("compat_result.set() requires a dict") + status = result.get("status") + if status not in VALID_STATUSES: + raise ValueError( + f"compat_result.set() status must be one of {sorted(VALID_STATUSES)}, " + f"got {status!r}" + ) + if status == "fail" and not result.get("error"): + raise ValueError("compat_result.set({'status': 'fail'}) requires 'error'") + if status == "not_applicable" and not result.get("reason"): + raise ValueError( + "compat_result.set({'status': 'not_applicable'}) requires 'reason'" + ) + self.value = dict(result) + + +@dataclass +class _CollectedResult: + feature_id: str + provider: str + nodeid: str + result: Dict[str, Any] + + +@dataclass +class _Collector: + items: List[_CollectedResult] = field(default_factory=list) + + +_COLLECTOR = _Collector() + + +@pytest.fixture +def compat_result() -> CompatResult: + """Per-test recorder for the (feature, provider) outcome. + + Tests should call `compat_result.set({"status": "pass"})` (or fail / + not_applicable) before returning. If a test exits without calling `.set()` + the harness records `status="fail"` with an explanatory error so that + every collected node maps to a real cell. + """ + return CompatResult() + + +def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]: + """Infer (feature_id, provider) from a test file path. + + Path shape: tests/claude_code//test_.py + Returns None if the file is not a per-feature test (e.g. unit tests + living under tests/claude_code/_driver_unit_tests/), so those don't + pollute the matrix artifact. + """ + name = node_path.name + if not name.startswith("test_") or not name.endswith(".py"): + return None + provider = name[len("test_") : -len(".py")] + feature_id = node_path.parent.name + if feature_id.startswith("_") or feature_id == "claude_code": + return None + return feature_id, provider + + +@pytest.hookimpl(hookwrapper=True) +def pytest_runtest_makereport(item, call): + """Capture compat_result.value at end-of-test and remember it for the artifact.""" + outcome = yield + report = outcome.get_result() + if report.when != "call": + return + + inferred = _infer_feature_and_provider(Path(str(item.path))) + if inferred is None: + return + feature_id, provider = inferred + + fixture = item.funcargs.get("compat_result") if hasattr(item, "funcargs") else None + reported: Optional[Dict[str, Any]] = getattr(fixture, "value", None) + + if reported is None: + if report.passed: + reported = { + "status": "fail", + "error": "test passed without calling compat_result.set(); " + "every compat test must report a status.", + } + else: + reported = { + "status": "fail", + "error": (str(report.longrepr) if report.longrepr else "test failed"), + } + + _COLLECTOR.items.append( + _CollectedResult( + feature_id=feature_id, + provider=provider, + nodeid=report.nodeid, + result=reported, + ) + ) + + +def pytest_sessionfinish(session, exitstatus): + """Write the structured results artifact at end of session.""" + artifact_path = os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH + payload = { + "schema_version": "1", + "results": [ + { + "feature_id": item.feature_id, + "provider": item.provider, + "nodeid": item.nodeid, + "result": item.result, + } + for item in _COLLECTOR.items + ], + } + Path(artifact_path).write_text(json.dumps(payload, indent=2, sort_keys=True)) diff --git a/tests/claude_code/manifest.yaml b/tests/claude_code/manifest.yaml new file mode 100644 index 00000000000..a9ae1e0d9fe --- /dev/null +++ b/tests/claude_code/manifest.yaml @@ -0,0 +1,26 @@ +# Claude Code Compatibility Matrix — feature manifest. +# +# Defines the row order of the matrix and maps each feature_id to its +# human-readable display name. Adding a new feature to the matrix is a +# three-step change: +# 1. Append an entry to `features:` below. +# 2. Create a directory `tests/claude_code//`. +# 3. Add per-provider test files inside that directory. +# +# `feature_id` MUST match the directory name on disk; the test harness +# infers (feature, provider) for each test from its file path. + +schema_version: "1" + +# Provider column order in the rendered matrix. +providers: + - anthropic + - bedrock_invoke + - bedrock_converse + - vertex_ai + - azure + +# Feature row order. +features: + - id: basic_messaging_non_streaming + name: Basic messaging (non-streaming) diff --git a/tests/claude_code/matrix_builder.py b/tests/claude_code/matrix_builder.py new file mode 100644 index 00000000000..b9bf8ecc598 --- /dev/null +++ b/tests/claude_code/matrix_builder.py @@ -0,0 +1,179 @@ +"""Matrix JSON Builder. + +Pure-function module that consumes the pytest-produced `compat-results.json`, +the manifest, and run metadata, and emits the final `compatibility-matrix.json` +conforming to the schema published in the PRD. + +This module is deliberately free of subprocess, network, or filesystem side +effects in its public API — the public entry points take pre-loaded inputs +and return data structures, so they can be exercised by golden-file tests +without I/O. A small `build_from_paths()` convenience wrapper does the I/O +for callers that need it (the daily-cron publisher). +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Dict, List, Mapping, Optional, Sequence + +import yaml + +SCHEMA_VERSION = "1" +VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"} + + +class ManifestError(ValueError): + """Raised when `manifest.yaml` is malformed.""" + + +class ResultsError(ValueError): + """Raised when the pytest results artifact is malformed.""" + + +def load_manifest(path: Path) -> Dict[str, Any]: + """Load and validate `manifest.yaml`. + + Returns a dict with keys: schema_version, providers, features. Raises + ManifestError on missing fields or schema mismatch. + """ + raw = yaml.safe_load(path.read_text()) + if not isinstance(raw, dict): + raise ManifestError(f"manifest at {path} is not a mapping") + schema_version = str(raw.get("schema_version", "")) + if schema_version != SCHEMA_VERSION: + raise ManifestError( + f"manifest schema_version {schema_version!r} does not match " + f"builder version {SCHEMA_VERSION!r}" + ) + providers = raw.get("providers") + if not isinstance(providers, list) or not providers: + raise ManifestError("manifest.providers must be a non-empty list") + features = raw.get("features") + if not isinstance(features, list) or not features: + raise ManifestError("manifest.features must be a non-empty list") + for feature in features: + if not isinstance(feature, dict): + raise ManifestError("each feature must be a mapping") + if not feature.get("id") or not feature.get("name"): + raise ManifestError("each feature must have id and name") + return raw + + +def load_results(path: Path) -> List[Dict[str, Any]]: + """Load the pytest results artifact and return its `results` list.""" + raw = json.loads(path.read_text()) + if not isinstance(raw, dict) or not isinstance(raw.get("results"), list): + raise ResultsError(f"results artifact at {path} has no `results` list") + return raw["results"] + + +def build_matrix( + *, + manifest: Mapping[str, Any], + results: Sequence[Mapping[str, Any]], + litellm_version: str, + claude_code_version: str, + generated_at: str, +) -> Dict[str, Any]: + """Build the published matrix JSON from pre-loaded inputs. + + Empty cells (no test ran for a (feature, provider) and no + `not_applicable` was declared) are filled in with `not_tested`. If + multiple results report on the same cell — e.g. a per-feature test + file containing one parametrize per Claude model — the cell aggregates + to `pass` only if every model passed; otherwise `fail` with the first + breaking model surfaced in the error. + """ + providers: List[str] = list(manifest["providers"]) + feature_specs: List[Dict[str, Any]] = list(manifest["features"]) + + grouped: Dict[tuple, List[Dict[str, Any]]] = {} + for entry in results: + if not isinstance(entry, Mapping): + continue + feature_id = entry.get("feature_id") + provider = entry.get("provider") + result = entry.get("result") + if not feature_id or not provider or not isinstance(result, Mapping): + continue + if result.get("status") not in VALID_STATUSES: + continue + grouped.setdefault((feature_id, provider), []).append(dict(result)) + + features_out: List[Dict[str, Any]] = [] + for spec in feature_specs: + feature_id = spec["id"] + cells: Dict[str, Dict[str, Any]] = {} + for provider in providers: + cell_results = grouped.get((feature_id, provider), []) + cells[provider] = _aggregate_cell(cell_results) + features_out.append( + { + "id": feature_id, + "name": spec["name"], + "providers": cells, + } + ) + + return { + "schema_version": SCHEMA_VERSION, + "generated_at": generated_at, + "litellm_version": litellm_version, + "claude_code_version": claude_code_version, + "providers": providers, + "features": features_out, + } + + +def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]: + """Aggregate a list of per-model results into a single cell status. + + Order of precedence (most informative wins): + - Any `fail` → cell is `fail` with the first failure's error. + - `not_applicable` → cell is `not_applicable` with the reason. + - `pass` → cell is `pass`. + - empty / nothing recognized → `not_tested`. + """ + if not results: + return {"status": "not_tested"} + + for r in results: + if r.get("status") == "fail": + return {"status": "fail", "error": str(r.get("error", "test failed"))} + + for r in results: + if r.get("status") == "not_applicable": + return { + "status": "not_applicable", + "reason": str(r.get("reason", "not applicable")), + } + + if all(r.get("status") == "pass" for r in results): + return {"status": "pass"} + + return {"status": "not_tested"} + + +def build_from_paths( + *, + manifest_path: Path, + results_path: Path, + litellm_version: str, + claude_code_version: str, + generated_at: str, + output_path: Optional[Path] = None, +) -> Dict[str, Any]: + """I/O wrapper around build_matrix used by the publisher script.""" + manifest = load_manifest(manifest_path) + results = load_results(results_path) + matrix = build_matrix( + manifest=manifest, + results=results, + litellm_version=litellm_version, + claude_code_version=claude_code_version, + generated_at=generated_at, + ) + if output_path is not None: + output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n") + return matrix diff --git a/tests/claude_code/sample_compatibility-matrix.json b/tests/claude_code/sample_compatibility-matrix.json new file mode 100644 index 00000000000..d4e38818e8e --- /dev/null +++ b/tests/claude_code/sample_compatibility-matrix.json @@ -0,0 +1,36 @@ +{ + "schema_version": "1", + "generated_at": "2026-04-25T00:00:00Z", + "litellm_version": "v1.83.0-stable", + "claude_code_version": "2.1.120", + "providers": [ + "anthropic", + "bedrock_invoke", + "bedrock_converse", + "vertex_ai", + "azure" + ], + "features": [ + { + "id": "basic_messaging_non_streaming", + "name": "Basic messaging (non-streaming)", + "providers": { + "anthropic": { + "status": "pass" + }, + "bedrock_invoke": { + "status": "not_tested" + }, + "bedrock_converse": { + "status": "not_tested" + }, + "vertex_ai": { + "status": "not_tested" + }, + "azure": { + "status": "not_tested" + } + } + } + ] +}