From 0bf013f620888a3dd900c360e0fae8c1d507194c Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 25 Apr 2026 04:07:02 +0000
Subject: [PATCH] RALPH: tracer-bullet for Claude Code compatibility matrix
(#26477, PRD #26476)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Slice 1 of the Claude Code Compatibility Matrix: the thinnest end-to-end
path through every layer for a single (feature, provider) cell, so a
future docs page can render a real green cell sourced from a real test.
What landed in this repo:
- tests/claude_code/manifest.yaml — feature manifest with one entry
(basic_messaging_non_streaming) plus the v0 provider column order.
- tests/claude_code/cli_driver.py — Claude Code CLI Driver. One entry
point (run_claude); handles subprocess assembly, env overlay, stream-JSON
parsing, and structured failure modes. `runner=` is a unit-test seam.
- tests/claude_code/conftest.py — `compat_result` fixture (tagged-union
recorder) + pytest_runtest_makereport hook that infers (feature, provider)
from the file path and writes a structured compat-results.json artifact.
- tests/claude_code/basic_messaging_non_streaming/test_anthropic.py — the
one cell, parametrized over Haiku/Sonnet/Opus per the PRD's per-cell
model rule.
- tests/claude_code/matrix_builder.py — pure-function builder from
(manifest, results, run-metadata) to the v1 JSON schema. Aggregates per-
model results into one cell (pass iff all pass). build_from_paths is the
thin I/O wrapper for the publisher.
- tests/claude_code/sample_compatibility-matrix.json — hand-authored sample
of the v1 JSON; copied to the docs repo by hand as part of this slice.
- Unit tests: 10 driver tests (mocked subprocess), 9 compat_result tests,
10 matrix-builder golden-file tests. 29/29 pass.
Key decisions:
- (feature, provider) is inferred from file path, not declared in metadata —
mirrors the PRD's "no drift" goal.
- Driver injects subprocess via a `runner` kwarg so unit tests don't need
the real `claude` CLI; production callers leave it default.
- Builder is a pure function on Mappings/Sequences; load/write live in a
thin `build_from_paths` wrapper. Golden-file tests pin the schema.
- `_driver_unit_tests/` and `_builder_unit_tests/` are prefixed with `_`
so the conftest's path-inference hook skips them and they don't
pollute the matrix artifact.
- `compat-results.json` added to .gitignore (CI-only output).
Out of scope per CLAUDE.md (docs live in BerriAI/litellm-docs):
- The MDX page `docs/tutorials/claude-code-compatibility` and the
`` React component. The hand-authored
compatibility-matrix.json (`sample_compatibility-matrix.json` in this
repo) is the artifact those docs files will consume; opening that doc
PR is the next step in this slice.
Co-Authored-By: Claude Opus 4.7
---
.gitignore | 5 +-
tests/claude_code/__init__.py | 0
.../_builder_unit_tests/__init__.py | 0
.../fixtures/expected_matrix.json | 38 +++
.../fixtures/manifest.yaml | 9 +
.../_builder_unit_tests/fixtures/results.json | 41 +++
.../test_matrix_builder.py | 192 ++++++++++++++
.../_driver_unit_tests/__init__.py | 0
.../_driver_unit_tests/test_cli_driver.py | 239 ++++++++++++++++++
.../_driver_unit_tests/test_compat_result.py | 70 +++++
.../basic_messaging_non_streaming/__init__.py | 0
.../test_anthropic.py | 98 +++++++
tests/claude_code/cli_driver.py | 194 ++++++++++++++
tests/claude_code/conftest.py | 164 ++++++++++++
tests/claude_code/manifest.yaml | 26 ++
tests/claude_code/matrix_builder.py | 179 +++++++++++++
.../sample_compatibility-matrix.json | 36 +++
17 files changed, 1290 insertions(+), 1 deletion(-)
create mode 100644 tests/claude_code/__init__.py
create mode 100644 tests/claude_code/_builder_unit_tests/__init__.py
create mode 100644 tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json
create mode 100644 tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml
create mode 100644 tests/claude_code/_builder_unit_tests/fixtures/results.json
create mode 100644 tests/claude_code/_builder_unit_tests/test_matrix_builder.py
create mode 100644 tests/claude_code/_driver_unit_tests/__init__.py
create mode 100644 tests/claude_code/_driver_unit_tests/test_cli_driver.py
create mode 100644 tests/claude_code/_driver_unit_tests/test_compat_result.py
create mode 100644 tests/claude_code/basic_messaging_non_streaming/__init__.py
create mode 100644 tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
create mode 100644 tests/claude_code/cli_driver.py
create mode 100644 tests/claude_code/conftest.py
create mode 100644 tests/claude_code/manifest.yaml
create mode 100644 tests/claude_code/matrix_builder.py
create mode 100644 tests/claude_code/sample_compatibility-matrix.json
diff --git a/.gitignore b/.gitignore
index 38bf9554b5b..5f07ee4007c 100644
--- a/.gitignore
+++ b/.gitignore
@@ -100,4 +100,7 @@ STABILIZATION_TODO.md
**/test-results
**/playwright-report
**/*.storageState.json
-**/coverage
\ No newline at end of file
+**/coverage
+
+# Claude Code compatibility-matrix pytest artifact (CI-only output).
+compat-results.json
\ No newline at end of file
diff --git a/tests/claude_code/__init__.py b/tests/claude_code/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/tests/claude_code/_builder_unit_tests/__init__.py b/tests/claude_code/_builder_unit_tests/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json b/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json
new file mode 100644
index 00000000000..d3aca0142dc
--- /dev/null
+++ b/tests/claude_code/_builder_unit_tests/fixtures/expected_matrix.json
@@ -0,0 +1,38 @@
+{
+ "schema_version": "1",
+ "generated_at": "2026-04-25T00:00:00Z",
+ "litellm_version": "v1.83.0-stable",
+ "claude_code_version": "2.1.120",
+ "providers": [
+ "anthropic",
+ "bedrock_invoke"
+ ],
+ "features": [
+ {
+ "id": "basic_messaging_non_streaming",
+ "name": "Basic messaging (non-streaming)",
+ "providers": {
+ "anthropic": {
+ "status": "pass"
+ },
+ "bedrock_invoke": {
+ "status": "not_tested"
+ }
+ }
+ },
+ {
+ "id": "tool_use",
+ "name": "Tool use",
+ "providers": {
+ "anthropic": {
+ "status": "fail",
+ "error": "[claude-sonnet-4-6] tool call dropped"
+ },
+ "bedrock_invoke": {
+ "status": "not_applicable",
+ "reason": "tool use not yet wired up for Bedrock Invoke"
+ }
+ }
+ }
+ ]
+}
diff --git a/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml b/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml
new file mode 100644
index 00000000000..e88bdc6ddf5
--- /dev/null
+++ b/tests/claude_code/_builder_unit_tests/fixtures/manifest.yaml
@@ -0,0 +1,9 @@
+schema_version: "1"
+providers:
+ - anthropic
+ - bedrock_invoke
+features:
+ - id: basic_messaging_non_streaming
+ name: Basic messaging (non-streaming)
+ - id: tool_use
+ name: Tool use
diff --git a/tests/claude_code/_builder_unit_tests/fixtures/results.json b/tests/claude_code/_builder_unit_tests/fixtures/results.json
new file mode 100644
index 00000000000..2ede64f167c
--- /dev/null
+++ b/tests/claude_code/_builder_unit_tests/fixtures/results.json
@@ -0,0 +1,41 @@
+{
+ "schema_version": "1",
+ "results": [
+ {
+ "feature_id": "basic_messaging_non_streaming",
+ "provider": "anthropic",
+ "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-haiku-4-5]",
+ "result": {"status": "pass"}
+ },
+ {
+ "feature_id": "basic_messaging_non_streaming",
+ "provider": "anthropic",
+ "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-sonnet-4-6]",
+ "result": {"status": "pass"}
+ },
+ {
+ "feature_id": "basic_messaging_non_streaming",
+ "provider": "anthropic",
+ "nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-opus-4-7]",
+ "result": {"status": "pass"}
+ },
+ {
+ "feature_id": "tool_use",
+ "provider": "anthropic",
+ "nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-haiku-4-5]",
+ "result": {"status": "pass"}
+ },
+ {
+ "feature_id": "tool_use",
+ "provider": "anthropic",
+ "nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-sonnet-4-6]",
+ "result": {"status": "fail", "error": "[claude-sonnet-4-6] tool call dropped"}
+ },
+ {
+ "feature_id": "tool_use",
+ "provider": "bedrock_invoke",
+ "nodeid": "tests/claude_code/tool_use/test_bedrock_invoke.py::test_x[claude-haiku-4-5]",
+ "result": {"status": "not_applicable", "reason": "tool use not yet wired up for Bedrock Invoke"}
+ }
+ ]
+}
diff --git a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py
new file mode 100644
index 00000000000..c032027e135
--- /dev/null
+++ b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py
@@ -0,0 +1,192 @@
+"""Golden-file tests for the Matrix JSON Builder.
+
+These tests fix the published JSON schema. The builder is a pure function
+from (manifest, results, metadata) → matrix dict, so we feed it a fixture
+input set and compare the produced dict to a checked-in expected output.
+
+Any schema drift — intentional or accidental — surfaces as a diff in PR
+review.
+"""
+
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import pytest
+
+from tests.claude_code.matrix_builder import (
+ ManifestError,
+ ResultsError,
+ build_from_paths,
+ build_matrix,
+ load_manifest,
+ load_results,
+)
+
+FIXTURES = Path(__file__).parent / "fixtures"
+
+
+def test_build_matrix_matches_golden_file(tmp_path):
+ manifest = load_manifest(FIXTURES / "manifest.yaml")
+ results = load_results(FIXTURES / "results.json")
+ matrix = build_matrix(
+ manifest=manifest,
+ results=results,
+ litellm_version="v1.83.0-stable",
+ claude_code_version="2.1.120",
+ generated_at="2026-04-25T00:00:00Z",
+ )
+ expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
+ assert matrix == expected
+
+
+def test_build_matrix_pass_requires_all_models_pass():
+ """Multiple results in one cell must all be pass for the cell to be pass."""
+ manifest = {
+ "schema_version": "1",
+ "providers": ["anthropic"],
+ "features": [{"id": "f", "name": "F"}],
+ }
+ results = [
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ ]
+ matrix = build_matrix(
+ manifest=manifest,
+ results=results,
+ litellm_version="v",
+ claude_code_version="c",
+ generated_at="t",
+ )
+ assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
+
+
+def test_build_matrix_any_fail_makes_cell_fail():
+ manifest = {
+ "schema_version": "1",
+ "providers": ["anthropic"],
+ "features": [{"id": "f", "name": "F"}],
+ }
+ results = [
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ {
+ "feature_id": "f",
+ "provider": "anthropic",
+ "result": {"status": "fail", "error": "[claude-opus-4-7] timeout"},
+ },
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ ]
+ matrix = build_matrix(
+ manifest=manifest,
+ results=results,
+ litellm_version="v",
+ claude_code_version="c",
+ generated_at="t",
+ )
+ cell = matrix["features"][0]["providers"]["anthropic"]
+ assert cell["status"] == "fail"
+ assert cell["error"] == "[claude-opus-4-7] timeout"
+
+
+def test_build_matrix_fills_not_tested_for_missing_cells():
+ manifest = {
+ "schema_version": "1",
+ "providers": ["anthropic", "azure"],
+ "features": [{"id": "f", "name": "F"}],
+ }
+ results = [
+ {"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
+ ]
+ matrix = build_matrix(
+ manifest=manifest,
+ results=results,
+ litellm_version="v",
+ claude_code_version="c",
+ generated_at="t",
+ )
+ cells = matrix["features"][0]["providers"]
+ assert cells["anthropic"] == {"status": "pass"}
+ assert cells["azure"] == {"status": "not_tested"}
+
+
+def test_build_matrix_preserves_provider_and_feature_order():
+ manifest = {
+ "schema_version": "1",
+ "providers": ["azure", "anthropic", "vertex_ai"],
+ "features": [
+ {"id": "z", "name": "Z"},
+ {"id": "a", "name": "A"},
+ ],
+ }
+ matrix = build_matrix(
+ manifest=manifest,
+ results=[],
+ litellm_version="v",
+ claude_code_version="c",
+ generated_at="t",
+ )
+ assert matrix["providers"] == ["azure", "anthropic", "vertex_ai"]
+ assert [f["id"] for f in matrix["features"]] == ["z", "a"]
+ assert list(matrix["features"][0]["providers"].keys()) == [
+ "azure",
+ "anthropic",
+ "vertex_ai",
+ ]
+
+
+def test_build_matrix_emits_schema_version_one():
+ manifest = {
+ "schema_version": "1",
+ "providers": ["anthropic"],
+ "features": [{"id": "f", "name": "F"}],
+ }
+ matrix = build_matrix(
+ manifest=manifest,
+ results=[],
+ litellm_version="v",
+ claude_code_version="c",
+ generated_at="t",
+ )
+ assert matrix["schema_version"] == "1"
+
+
+def test_load_manifest_rejects_wrong_schema_version(tmp_path):
+ bad = tmp_path / "manifest.yaml"
+ bad.write_text(
+ 'schema_version: "2"\nproviders: [anthropic]\nfeatures:\n - id: f\n name: F\n'
+ )
+ with pytest.raises(ManifestError, match="schema_version"):
+ load_manifest(bad)
+
+
+def test_load_manifest_rejects_empty_features(tmp_path):
+ bad = tmp_path / "manifest.yaml"
+ bad.write_text('schema_version: "1"\nproviders: [anthropic]\nfeatures: []\n')
+ with pytest.raises(ManifestError):
+ load_manifest(bad)
+
+
+def test_load_results_rejects_missing_results_key(tmp_path):
+ bad = tmp_path / "results.json"
+ bad.write_text(json.dumps({"schema_version": "1"}))
+ with pytest.raises(ResultsError):
+ load_results(bad)
+
+
+def test_build_from_paths_writes_output(tmp_path):
+ out = tmp_path / "compatibility-matrix.json"
+ matrix = build_from_paths(
+ manifest_path=FIXTURES / "manifest.yaml",
+ results_path=FIXTURES / "results.json",
+ litellm_version="v1.83.0-stable",
+ claude_code_version="2.1.120",
+ generated_at="2026-04-25T00:00:00Z",
+ output_path=out,
+ )
+ assert out.exists()
+ on_disk = json.loads(out.read_text())
+ assert on_disk == matrix
+ expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
+ assert on_disk == expected
diff --git a/tests/claude_code/_driver_unit_tests/__init__.py b/tests/claude_code/_driver_unit_tests/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/tests/claude_code/_driver_unit_tests/test_cli_driver.py b/tests/claude_code/_driver_unit_tests/test_cli_driver.py
new file mode 100644
index 00000000000..fbdea1a0aa4
--- /dev/null
+++ b/tests/claude_code/_driver_unit_tests/test_cli_driver.py
@@ -0,0 +1,239 @@
+"""Unit tests for the Claude Code CLI Driver.
+
+These tests mock the subprocess so they run anywhere — no network, no
+`claude` install, no API keys. They cover the behavior contract:
+argument assembly, environment overlay, stream-JSON parsing, exit-code
+plumbing, and the structured failure modes (CLI not found, timeout).
+"""
+
+from __future__ import annotations
+
+import json
+import subprocess
+from dataclasses import dataclass
+from typing import List, Optional
+
+import pytest
+
+from tests.claude_code.cli_driver import (
+ ClaudeCLIError,
+ DriverResult,
+ run_claude,
+)
+
+
+@dataclass
+class _Completed:
+ returncode: int = 0
+ stdout: str = ""
+ stderr: str = ""
+
+
+def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
+ captured = {}
+
+ def runner(cmd, env, capture_output, text, timeout, check):
+ captured["cmd"] = cmd
+ captured["env"] = env
+ captured["timeout"] = timeout
+ return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
+
+ return runner, captured
+
+
+def test_run_claude_assembles_command_correctly():
+ runner, captured = _make_runner(
+ stdout='{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}\n'
+ )
+ run_claude(
+ prompt="hello",
+ model="claude-haiku-4-5",
+ base_url="http://localhost:4000",
+ api_key="sk-test",
+ runner=runner,
+ )
+ cmd = captured["cmd"]
+ assert cmd[0] == "claude"
+ assert "--print" in cmd
+ assert "--output-format" in cmd
+ assert "stream-json" in cmd
+ assert "--model" in cmd
+ assert "claude-haiku-4-5" in cmd
+ assert cmd[-1] == "hello"
+
+
+def test_run_claude_overlays_proxy_env():
+ runner, captured = _make_runner(stdout="")
+ run_claude(
+ prompt="hi",
+ model="claude-opus-4-7",
+ base_url="http://proxy.example:4000",
+ api_key="sk-abc",
+ runner=runner,
+ )
+ env = captured["env"]
+ assert env["ANTHROPIC_BASE_URL"] == "http://proxy.example:4000"
+ assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-abc"
+
+
+def test_run_claude_extra_env_takes_precedence_over_os_environ(monkeypatch):
+ monkeypatch.setenv("FOO", "from-os")
+ runner, captured = _make_runner(stdout="")
+ run_claude(
+ prompt="hi",
+ model="claude-opus-4-7",
+ base_url="http://localhost",
+ api_key="sk-abc",
+ extra_env={"FOO": "from-arg"},
+ runner=runner,
+ )
+ assert captured["env"]["FOO"] == "from-arg"
+
+
+def test_run_claude_parses_stream_json_assistant_text():
+ events = [
+ {"type": "system", "session_id": "abc"},
+ {
+ "type": "assistant",
+ "message": {
+ "content": [
+ {"type": "text", "text": "Hello "},
+ {"type": "text", "text": "world"},
+ ]
+ },
+ },
+ {"type": "result", "usage": {"input_tokens": 10, "output_tokens": 2}},
+ ]
+ stdout = "\n".join(json.dumps(e) for e in events) + "\n"
+ runner, _ = _make_runner(stdout=stdout)
+ result = run_claude(
+ prompt="hi",
+ model="claude-haiku-4-5",
+ base_url="http://localhost",
+ api_key="sk-abc",
+ runner=runner,
+ )
+ assert isinstance(result, DriverResult)
+ assert result.text == "Hello world"
+ assert len(result.events) == 3
+ assert result.usage == {"input_tokens": 10, "output_tokens": 2}
+ assert result.exit_code == 0
+
+
+def test_run_claude_handles_string_message_content():
+ """Some CLI versions emit `message.content` as a plain string."""
+ stdout = (
+ json.dumps({"type": "assistant", "message": {"content": "bare text"}}) + "\n"
+ )
+ runner, _ = _make_runner(stdout=stdout)
+ result = run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+ assert result.text == "bare text"
+
+
+def test_run_claude_skips_malformed_lines():
+ stdout = (
+ "not-json\n"
+ + json.dumps(
+ {
+ "type": "assistant",
+ "message": {"content": [{"type": "text", "text": "x"}]},
+ }
+ )
+ + "\n"
+ + "{also-bad\n"
+ )
+ runner, _ = _make_runner(stdout=stdout)
+ result = run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+ assert result.text == "x"
+ assert len(result.events) == 1
+
+
+def test_run_claude_propagates_nonzero_exit_code():
+ runner, _ = _make_runner(stdout="", returncode=2, stderr="auth failed")
+ result = run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+ assert result.exit_code == 2
+ assert result.stderr == "auth failed"
+ assert result.text == ""
+
+
+def test_run_claude_raises_on_missing_cli():
+ def runner(*args, **kwargs):
+ raise FileNotFoundError(2, "no such file", "claude")
+
+ with pytest.raises(ClaudeCLIError, match="claude CLI not found"):
+ run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+
+
+def test_run_claude_raises_on_timeout():
+ def runner(*args, **kwargs):
+ raise subprocess.TimeoutExpired(cmd="claude", timeout=1)
+
+ with pytest.raises(ClaudeCLIError, match="timed out"):
+ run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ timeout=1,
+ runner=runner,
+ )
+
+
+def test_run_claude_validates_required_params():
+ runner, _ = _make_runner()
+ with pytest.raises(ValueError, match="prompt"):
+ run_claude(
+ prompt="",
+ model="m",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+ with pytest.raises(ValueError, match="model"):
+ run_claude(
+ prompt="hi",
+ model="",
+ base_url="http://x",
+ api_key="k",
+ runner=runner,
+ )
+ with pytest.raises(ValueError, match="base_url"):
+ run_claude(
+ prompt="hi",
+ model="m",
+ base_url="",
+ api_key="k",
+ runner=runner,
+ )
+ with pytest.raises(ValueError, match="api_key"):
+ run_claude(
+ prompt="hi",
+ model="m",
+ base_url="http://x",
+ api_key="",
+ runner=runner,
+ )
diff --git a/tests/claude_code/_driver_unit_tests/test_compat_result.py b/tests/claude_code/_driver_unit_tests/test_compat_result.py
new file mode 100644
index 00000000000..04e19bdc482
--- /dev/null
+++ b/tests/claude_code/_driver_unit_tests/test_compat_result.py
@@ -0,0 +1,70 @@
+"""Tests for the `compat_result` fixture's tagged-union validation.
+
+The conftest's `pytest_runtest_makereport` hook is exercised end-to-end by
+the matrix-builder golden-file tests (which consume a results.json that
+the harness would produce). Here we just test the input-validation
+contract on `CompatResult.set()`.
+"""
+
+from __future__ import annotations
+
+import pytest
+
+from tests.claude_code.conftest import CompatResult
+
+
+def test_set_pass_is_accepted():
+ r = CompatResult()
+ r.set({"status": "pass"})
+ assert r.value == {"status": "pass"}
+
+
+def test_set_fail_requires_error():
+ r = CompatResult()
+ with pytest.raises(ValueError, match="requires 'error'"):
+ r.set({"status": "fail"})
+
+
+def test_set_fail_with_error_is_accepted():
+ r = CompatResult()
+ r.set({"status": "fail", "error": "boom"})
+ assert r.value == {"status": "fail", "error": "boom"}
+
+
+def test_set_not_applicable_requires_reason():
+ r = CompatResult()
+ with pytest.raises(ValueError, match="requires 'reason'"):
+ r.set({"status": "not_applicable"})
+
+
+def test_set_not_applicable_with_reason_is_accepted():
+ r = CompatResult()
+ r.set({"status": "not_applicable", "reason": "Bedrock has no /thinking"})
+ assert r.value == {"status": "not_applicable", "reason": "Bedrock has no /thinking"}
+
+
+def test_set_not_tested_is_accepted():
+ r = CompatResult()
+ r.set({"status": "not_tested"})
+ assert r.value == {"status": "not_tested"}
+
+
+def test_set_rejects_unknown_status():
+ r = CompatResult()
+ with pytest.raises(ValueError, match="status must be one of"):
+ r.set({"status": "maybe"})
+
+
+def test_set_rejects_non_dict():
+ r = CompatResult()
+ with pytest.raises(TypeError):
+ r.set("pass") # type: ignore[arg-type]
+
+
+def test_set_copies_input():
+ """Mutating the dict after set() must not change the stored value."""
+ r = CompatResult()
+ payload = {"status": "fail", "error": "x"}
+ r.set(payload)
+ payload["error"] = "mutated"
+ assert r.value["error"] == "x"
diff --git a/tests/claude_code/basic_messaging_non_streaming/__init__.py b/tests/claude_code/basic_messaging_non_streaming/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py b/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
new file mode 100644
index 00000000000..7e0a1f16583
--- /dev/null
+++ b/tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
@@ -0,0 +1,98 @@
+"""basic_messaging_non_streaming × Anthropic.
+
+The thinnest end-to-end path through every layer of the matrix: drive the
+real `claude` CLI in headless mode against a running LiteLLM proxy that
+routes to Anthropic, and report the outcome via `compat_result`.
+
+The (feature, provider) for this cell is inferred from the file path by
+`tests/claude_code/conftest.py`:
+
+ tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
+ feature_id provider
+
+Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
+4.7; the cell only goes green if all three pass. We parametrize over the
+three models and the conftest aggregator produces one cell from the three
+results.
+"""
+
+from __future__ import annotations
+
+import os
+
+import pytest
+
+from tests.claude_code.cli_driver import ClaudeCLIError, run_claude
+
+PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
+PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
+
+# Per the PRD: each cell is exercised against three Claude tiers via the
+# Anthropic provider. Aliases are configured in the LiteLLM proxy's
+# routing config; the driver only sends the alias.
+ANTHROPIC_MODELS = [
+ "claude-haiku-4-5",
+ "claude-sonnet-4-6",
+ "claude-opus-4-7",
+]
+
+
+@pytest.mark.parametrize("model", ANTHROPIC_MODELS)
+def test_basic_messaging_non_streaming_anthropic(compat_result, model):
+ """Drive the `claude` CLI against the LiteLLM proxy and assert a reply.
+
+ "Basic messaging" means: send a single user prompt, receive any
+ non-empty assistant text reply, no tools, no streaming, no thinking.
+ The whole point of this slice is to prove the path works at all —
+ so the assertion is intentionally lenient on the reply contents.
+ """
+ base_url = os.environ.get(PROXY_BASE_URL_ENV)
+ api_key = os.environ.get(PROXY_API_KEY_ENV)
+ if not base_url or not api_key:
+ compat_result.set(
+ {
+ "status": "fail",
+ "error": (
+ f"missing required env: set {PROXY_BASE_URL_ENV} and "
+ f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
+ ),
+ }
+ )
+ pytest.fail(
+ f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
+ )
+
+ try:
+ result = run_claude(
+ prompt="Reply with the single word 'pong' and nothing else.",
+ model=model,
+ base_url=base_url,
+ api_key=api_key,
+ )
+ except ClaudeCLIError as exc:
+ compat_result.set({"status": "fail", "error": f"[{model}] {exc}"})
+ pytest.fail(str(exc), pytrace=False)
+ return
+
+ if result.exit_code != 0:
+ compat_result.set(
+ {
+ "status": "fail",
+ "error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}",
+ }
+ )
+ pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False)
+ return
+
+ if not result.text.strip():
+ compat_result.set(
+ {
+ "status": "fail",
+ "error": f"[{model}] claude returned empty assistant text",
+ }
+ )
+ pytest.fail(f"empty reply for {model}", pytrace=False)
+ return
+
+ compat_result.set({"status": "pass"})
diff --git a/tests/claude_code/cli_driver.py b/tests/claude_code/cli_driver.py
new file mode 100644
index 00000000000..cd8732491d3
--- /dev/null
+++ b/tests/claude_code/cli_driver.py
@@ -0,0 +1,194 @@
+"""Claude Code CLI Driver.
+
+A thin wrapper around the `claude` CLI in headless mode. Every compatibility
+test consumes only this module — tests must never shell out directly. This
+keeps the subprocess assembly, stream-JSON parsing, and result shape in a
+single place that can be unit-tested with a mocked subprocess.
+
+The driver is deliberately small: it knows how to invoke the CLI, drain its
+stream-JSON output, and return a structured `DriverResult`. Higher-level
+matrix concerns (status aggregation, manifest lookup, JSON serialization)
+live in `matrix_builder.py`.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import subprocess
+from dataclasses import dataclass, field
+from typing import Any, Dict, List, Mapping, Optional, Sequence
+
+CLAUDE_CLI_DEFAULT = "claude"
+DEFAULT_TIMEOUT_SECONDS = 120
+
+
+class ClaudeCLIError(RuntimeError):
+ """Raised when the `claude` CLI cannot be invoked or returns a fatal error."""
+
+
+@dataclass
+class DriverResult:
+ """Structured outcome of a single `claude` CLI invocation.
+
+ `text` is the assistant's final user-visible reply (joined across any
+ intermediate `assistant` events for non-streaming runs). `events` is the
+ raw list of stream-JSON objects emitted by the CLI, preserved so test
+ authors can write feature-specific assertions (tool calls, cache hits,
+ usage) without re-parsing stdout.
+ """
+
+ text: str
+ events: List[Dict[str, Any]] = field(default_factory=list)
+ exit_code: int = 0
+ stderr: str = ""
+ usage: Optional[Dict[str, Any]] = None
+ duration_ms: Optional[int] = None
+
+
+def run_claude(
+ *,
+ prompt: str,
+ model: str,
+ base_url: str,
+ api_key: str,
+ extra_env: Optional[Mapping[str, str]] = None,
+ extra_args: Optional[Sequence[str]] = None,
+ cli_path: str = CLAUDE_CLI_DEFAULT,
+ timeout: float = DEFAULT_TIMEOUT_SECONDS,
+ runner: Optional[Any] = None,
+) -> DriverResult:
+ """Invoke `claude` once in headless stream-JSON mode and return the result.
+
+ The CLI is pointed at a LiteLLM proxy via `ANTHROPIC_BASE_URL` /
+ `ANTHROPIC_AUTH_TOKEN`, so the same code path exercises every provider
+ column — only the model id and the proxy's routing differ between
+ invocations.
+
+ `runner` is an injection seam used by the unit tests: by default we call
+ `subprocess.run`, but the test suite swaps in a fake that yields canned
+ stream-JSON. Production callers should never set it.
+ """
+ if not prompt:
+ raise ValueError("prompt must be a non-empty string")
+ if not model:
+ raise ValueError("model must be a non-empty string")
+ if not base_url:
+ raise ValueError("base_url must be a non-empty string")
+ if not api_key:
+ raise ValueError("api_key must be a non-empty string")
+
+ cmd: List[str] = [
+ cli_path,
+ "--print",
+ "--output-format",
+ "stream-json",
+ "--verbose",
+ "--model",
+ model,
+ prompt,
+ ]
+ if extra_args:
+ cmd.extend(extra_args)
+
+ env = {**os.environ, **(extra_env or {})}
+ env["ANTHROPIC_BASE_URL"] = base_url
+ env["ANTHROPIC_AUTH_TOKEN"] = api_key
+
+ run_fn = runner or subprocess.run
+ try:
+ completed = run_fn(
+ cmd,
+ env=env,
+ capture_output=True,
+ text=True,
+ timeout=timeout,
+ check=False,
+ )
+ except FileNotFoundError as exc:
+ raise ClaudeCLIError(
+ f"claude CLI not found at {cli_path!r}; install with `npm i -g @anthropic-ai/claude-code`"
+ ) from exc
+ except subprocess.TimeoutExpired as exc:
+ raise ClaudeCLIError(f"claude CLI timed out after {timeout}s") from exc
+
+ events = _parse_stream_json(completed.stdout or "")
+ text = _extract_assistant_text(events)
+ usage = _extract_usage(events)
+
+ return DriverResult(
+ text=text,
+ events=events,
+ exit_code=completed.returncode,
+ stderr=completed.stderr or "",
+ usage=usage,
+ )
+
+
+def _parse_stream_json(stdout: str) -> List[Dict[str, Any]]:
+ """Parse newline-delimited JSON emitted by `claude --output-format stream-json`.
+
+ Lines that don't parse as JSON are silently skipped — the CLI occasionally
+ emits debug output we don't care about, and a single malformed line should
+ not abort the whole run. Real failure modes surface via exit code.
+ """
+ events: List[Dict[str, Any]] = []
+ for line in stdout.splitlines():
+ line = line.strip()
+ if not line:
+ continue
+ try:
+ obj = json.loads(line)
+ except json.JSONDecodeError:
+ continue
+ if isinstance(obj, dict):
+ events.append(obj)
+ return events
+
+
+def _extract_assistant_text(events: Sequence[Mapping[str, Any]]) -> str:
+ """Concatenate the text content of every `assistant` event in order.
+
+ The non-streaming `--print` path emits a single `assistant` event whose
+ `message.content` is a list of content blocks. We walk the blocks and
+ join every `text` block — the CLI prints other block types (e.g.
+ `tool_use`) which we ignore for the basic-messaging case.
+ """
+ chunks: List[str] = []
+ for event in events:
+ if event.get("type") != "assistant":
+ continue
+ message = event.get("message") or {}
+ content = message.get("content")
+ if isinstance(content, str):
+ chunks.append(content)
+ continue
+ if not isinstance(content, list):
+ continue
+ for block in content:
+ if not isinstance(block, dict):
+ continue
+ if block.get("type") == "text" and isinstance(block.get("text"), str):
+ chunks.append(block["text"])
+ return "".join(chunks)
+
+
+def _extract_usage(events: Sequence[Mapping[str, Any]]) -> Optional[Dict[str, Any]]:
+ """Return the most recent `usage` block seen on any event, if any.
+
+ The CLI surfaces token + cache usage on the final `result` event for
+ non-streaming runs, but earlier events also carry partial usage in some
+ versions; taking the last non-empty one is the safe default.
+ """
+ last: Optional[Dict[str, Any]] = None
+ for event in events:
+ usage = event.get("usage")
+ if isinstance(usage, dict) and usage:
+ last = usage
+ continue
+ message = event.get("message")
+ if isinstance(message, dict):
+ inner = message.get("usage")
+ if isinstance(inner, dict) and inner:
+ last = inner
+ return last
diff --git a/tests/claude_code/conftest.py b/tests/claude_code/conftest.py
new file mode 100644
index 00000000000..567c24720ed
--- /dev/null
+++ b/tests/claude_code/conftest.py
@@ -0,0 +1,164 @@
+"""Pytest plumbing for the Claude Code compatibility matrix.
+
+Two responsibilities live here:
+
+1. The `compat_result` fixture — the only API a test author needs to learn.
+ Tests call `compat_result.set({"status": "pass"})` (or fail / not_applicable)
+ to report their outcome as a tagged union. The fixture is per-test and
+ stores the last value reported.
+
+2. The `pytest_runtest_makereport` hook — captures each test's reported result,
+ infers (feature, provider) from the file path, and writes a single
+ `compat-results.json` artifact next to JUnit XML. The Matrix JSON Builder
+ consumes this artifact to produce the published `compatibility-matrix.json`.
+
+The (feature, provider) inference comes from the test file path: the parent
+directory name is the feature_id (matching `manifest.yaml`), and the file
+stem after the leading `test_` is the provider id. This avoids per-file
+metadata that drifts.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any, Dict, List, Optional
+
+import pytest
+
+VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
+RESULTS_ARTIFACT_ENV = "COMPAT_RESULTS_PATH"
+DEFAULT_ARTIFACT_PATH = "compat-results.json"
+
+
+@dataclass
+class CompatResult:
+ """Per-test recorder for compatibility outcomes.
+
+ Tests interact only via `.set(...)`. `.value` is read by the
+ `pytest_runtest_makereport` hook after the test body finishes.
+ """
+
+ value: Optional[Dict[str, Any]] = None
+
+ def set(self, result: Dict[str, Any]) -> None:
+ if not isinstance(result, dict):
+ raise TypeError("compat_result.set() requires a dict")
+ status = result.get("status")
+ if status not in VALID_STATUSES:
+ raise ValueError(
+ f"compat_result.set() status must be one of {sorted(VALID_STATUSES)}, "
+ f"got {status!r}"
+ )
+ if status == "fail" and not result.get("error"):
+ raise ValueError("compat_result.set({'status': 'fail'}) requires 'error'")
+ if status == "not_applicable" and not result.get("reason"):
+ raise ValueError(
+ "compat_result.set({'status': 'not_applicable'}) requires 'reason'"
+ )
+ self.value = dict(result)
+
+
+@dataclass
+class _CollectedResult:
+ feature_id: str
+ provider: str
+ nodeid: str
+ result: Dict[str, Any]
+
+
+@dataclass
+class _Collector:
+ items: List[_CollectedResult] = field(default_factory=list)
+
+
+_COLLECTOR = _Collector()
+
+
+@pytest.fixture
+def compat_result() -> CompatResult:
+ """Per-test recorder for the (feature, provider) outcome.
+
+ Tests should call `compat_result.set({"status": "pass"})` (or fail /
+ not_applicable) before returning. If a test exits without calling `.set()`
+ the harness records `status="fail"` with an explanatory error so that
+ every collected node maps to a real cell.
+ """
+ return CompatResult()
+
+
+def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]:
+ """Infer (feature_id, provider) from a test file path.
+
+ Path shape: tests/claude_code//test_.py
+ Returns None if the file is not a per-feature test (e.g. unit tests
+ living under tests/claude_code/_driver_unit_tests/), so those don't
+ pollute the matrix artifact.
+ """
+ name = node_path.name
+ if not name.startswith("test_") or not name.endswith(".py"):
+ return None
+ provider = name[len("test_") : -len(".py")]
+ feature_id = node_path.parent.name
+ if feature_id.startswith("_") or feature_id == "claude_code":
+ return None
+ return feature_id, provider
+
+
+@pytest.hookimpl(hookwrapper=True)
+def pytest_runtest_makereport(item, call):
+ """Capture compat_result.value at end-of-test and remember it for the artifact."""
+ outcome = yield
+ report = outcome.get_result()
+ if report.when != "call":
+ return
+
+ inferred = _infer_feature_and_provider(Path(str(item.path)))
+ if inferred is None:
+ return
+ feature_id, provider = inferred
+
+ fixture = item.funcargs.get("compat_result") if hasattr(item, "funcargs") else None
+ reported: Optional[Dict[str, Any]] = getattr(fixture, "value", None)
+
+ if reported is None:
+ if report.passed:
+ reported = {
+ "status": "fail",
+ "error": "test passed without calling compat_result.set(); "
+ "every compat test must report a status.",
+ }
+ else:
+ reported = {
+ "status": "fail",
+ "error": (str(report.longrepr) if report.longrepr else "test failed"),
+ }
+
+ _COLLECTOR.items.append(
+ _CollectedResult(
+ feature_id=feature_id,
+ provider=provider,
+ nodeid=report.nodeid,
+ result=reported,
+ )
+ )
+
+
+def pytest_sessionfinish(session, exitstatus):
+ """Write the structured results artifact at end of session."""
+ artifact_path = os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH
+ payload = {
+ "schema_version": "1",
+ "results": [
+ {
+ "feature_id": item.feature_id,
+ "provider": item.provider,
+ "nodeid": item.nodeid,
+ "result": item.result,
+ }
+ for item in _COLLECTOR.items
+ ],
+ }
+ Path(artifact_path).write_text(json.dumps(payload, indent=2, sort_keys=True))
diff --git a/tests/claude_code/manifest.yaml b/tests/claude_code/manifest.yaml
new file mode 100644
index 00000000000..a9ae1e0d9fe
--- /dev/null
+++ b/tests/claude_code/manifest.yaml
@@ -0,0 +1,26 @@
+# Claude Code Compatibility Matrix — feature manifest.
+#
+# Defines the row order of the matrix and maps each feature_id to its
+# human-readable display name. Adding a new feature to the matrix is a
+# three-step change:
+# 1. Append an entry to `features:` below.
+# 2. Create a directory `tests/claude_code//`.
+# 3. Add per-provider test files inside that directory.
+#
+# `feature_id` MUST match the directory name on disk; the test harness
+# infers (feature, provider) for each test from its file path.
+
+schema_version: "1"
+
+# Provider column order in the rendered matrix.
+providers:
+ - anthropic
+ - bedrock_invoke
+ - bedrock_converse
+ - vertex_ai
+ - azure
+
+# Feature row order.
+features:
+ - id: basic_messaging_non_streaming
+ name: Basic messaging (non-streaming)
diff --git a/tests/claude_code/matrix_builder.py b/tests/claude_code/matrix_builder.py
new file mode 100644
index 00000000000..b9bf8ecc598
--- /dev/null
+++ b/tests/claude_code/matrix_builder.py
@@ -0,0 +1,179 @@
+"""Matrix JSON Builder.
+
+Pure-function module that consumes the pytest-produced `compat-results.json`,
+the manifest, and run metadata, and emits the final `compatibility-matrix.json`
+conforming to the schema published in the PRD.
+
+This module is deliberately free of subprocess, network, or filesystem side
+effects in its public API — the public entry points take pre-loaded inputs
+and return data structures, so they can be exercised by golden-file tests
+without I/O. A small `build_from_paths()` convenience wrapper does the I/O
+for callers that need it (the daily-cron publisher).
+"""
+
+from __future__ import annotations
+
+import json
+from pathlib import Path
+from typing import Any, Dict, List, Mapping, Optional, Sequence
+
+import yaml
+
+SCHEMA_VERSION = "1"
+VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
+
+
+class ManifestError(ValueError):
+ """Raised when `manifest.yaml` is malformed."""
+
+
+class ResultsError(ValueError):
+ """Raised when the pytest results artifact is malformed."""
+
+
+def load_manifest(path: Path) -> Dict[str, Any]:
+ """Load and validate `manifest.yaml`.
+
+ Returns a dict with keys: schema_version, providers, features. Raises
+ ManifestError on missing fields or schema mismatch.
+ """
+ raw = yaml.safe_load(path.read_text())
+ if not isinstance(raw, dict):
+ raise ManifestError(f"manifest at {path} is not a mapping")
+ schema_version = str(raw.get("schema_version", ""))
+ if schema_version != SCHEMA_VERSION:
+ raise ManifestError(
+ f"manifest schema_version {schema_version!r} does not match "
+ f"builder version {SCHEMA_VERSION!r}"
+ )
+ providers = raw.get("providers")
+ if not isinstance(providers, list) or not providers:
+ raise ManifestError("manifest.providers must be a non-empty list")
+ features = raw.get("features")
+ if not isinstance(features, list) or not features:
+ raise ManifestError("manifest.features must be a non-empty list")
+ for feature in features:
+ if not isinstance(feature, dict):
+ raise ManifestError("each feature must be a mapping")
+ if not feature.get("id") or not feature.get("name"):
+ raise ManifestError("each feature must have id and name")
+ return raw
+
+
+def load_results(path: Path) -> List[Dict[str, Any]]:
+ """Load the pytest results artifact and return its `results` list."""
+ raw = json.loads(path.read_text())
+ if not isinstance(raw, dict) or not isinstance(raw.get("results"), list):
+ raise ResultsError(f"results artifact at {path} has no `results` list")
+ return raw["results"]
+
+
+def build_matrix(
+ *,
+ manifest: Mapping[str, Any],
+ results: Sequence[Mapping[str, Any]],
+ litellm_version: str,
+ claude_code_version: str,
+ generated_at: str,
+) -> Dict[str, Any]:
+ """Build the published matrix JSON from pre-loaded inputs.
+
+ Empty cells (no test ran for a (feature, provider) and no
+ `not_applicable` was declared) are filled in with `not_tested`. If
+ multiple results report on the same cell — e.g. a per-feature test
+ file containing one parametrize per Claude model — the cell aggregates
+ to `pass` only if every model passed; otherwise `fail` with the first
+ breaking model surfaced in the error.
+ """
+ providers: List[str] = list(manifest["providers"])
+ feature_specs: List[Dict[str, Any]] = list(manifest["features"])
+
+ grouped: Dict[tuple, List[Dict[str, Any]]] = {}
+ for entry in results:
+ if not isinstance(entry, Mapping):
+ continue
+ feature_id = entry.get("feature_id")
+ provider = entry.get("provider")
+ result = entry.get("result")
+ if not feature_id or not provider or not isinstance(result, Mapping):
+ continue
+ if result.get("status") not in VALID_STATUSES:
+ continue
+ grouped.setdefault((feature_id, provider), []).append(dict(result))
+
+ features_out: List[Dict[str, Any]] = []
+ for spec in feature_specs:
+ feature_id = spec["id"]
+ cells: Dict[str, Dict[str, Any]] = {}
+ for provider in providers:
+ cell_results = grouped.get((feature_id, provider), [])
+ cells[provider] = _aggregate_cell(cell_results)
+ features_out.append(
+ {
+ "id": feature_id,
+ "name": spec["name"],
+ "providers": cells,
+ }
+ )
+
+ return {
+ "schema_version": SCHEMA_VERSION,
+ "generated_at": generated_at,
+ "litellm_version": litellm_version,
+ "claude_code_version": claude_code_version,
+ "providers": providers,
+ "features": features_out,
+ }
+
+
+def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
+ """Aggregate a list of per-model results into a single cell status.
+
+ Order of precedence (most informative wins):
+ - Any `fail` → cell is `fail` with the first failure's error.
+ - `not_applicable` → cell is `not_applicable` with the reason.
+ - `pass` → cell is `pass`.
+ - empty / nothing recognized → `not_tested`.
+ """
+ if not results:
+ return {"status": "not_tested"}
+
+ for r in results:
+ if r.get("status") == "fail":
+ return {"status": "fail", "error": str(r.get("error", "test failed"))}
+
+ for r in results:
+ if r.get("status") == "not_applicable":
+ return {
+ "status": "not_applicable",
+ "reason": str(r.get("reason", "not applicable")),
+ }
+
+ if all(r.get("status") == "pass" for r in results):
+ return {"status": "pass"}
+
+ return {"status": "not_tested"}
+
+
+def build_from_paths(
+ *,
+ manifest_path: Path,
+ results_path: Path,
+ litellm_version: str,
+ claude_code_version: str,
+ generated_at: str,
+ output_path: Optional[Path] = None,
+) -> Dict[str, Any]:
+ """I/O wrapper around build_matrix used by the publisher script."""
+ manifest = load_manifest(manifest_path)
+ results = load_results(results_path)
+ matrix = build_matrix(
+ manifest=manifest,
+ results=results,
+ litellm_version=litellm_version,
+ claude_code_version=claude_code_version,
+ generated_at=generated_at,
+ )
+ if output_path is not None:
+ output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n")
+ return matrix
diff --git a/tests/claude_code/sample_compatibility-matrix.json b/tests/claude_code/sample_compatibility-matrix.json
new file mode 100644
index 00000000000..d4e38818e8e
--- /dev/null
+++ b/tests/claude_code/sample_compatibility-matrix.json
@@ -0,0 +1,36 @@
+{
+ "schema_version": "1",
+ "generated_at": "2026-04-25T00:00:00Z",
+ "litellm_version": "v1.83.0-stable",
+ "claude_code_version": "2.1.120",
+ "providers": [
+ "anthropic",
+ "bedrock_invoke",
+ "bedrock_converse",
+ "vertex_ai",
+ "azure"
+ ],
+ "features": [
+ {
+ "id": "basic_messaging_non_streaming",
+ "name": "Basic messaging (non-streaming)",
+ "providers": {
+ "anthropic": {
+ "status": "pass"
+ },
+ "bedrock_invoke": {
+ "status": "not_tested"
+ },
+ "bedrock_converse": {
+ "status": "not_tested"
+ },
+ "vertex_ai": {
+ "status": "not_tested"
+ },
+ "azure": {
+ "status": "not_tested"
+ }
+ }
+ }
+ ]
+}