RALPH: tracer-bullet for Claude Code compatibility matrix (#26477, PRD #26476)

Slice 1 of the Claude Code Compatibility Matrix: the thinnest end-to-end
path through every layer for a single (feature, provider) cell, so a
future docs page can render a real green cell sourced from a real test.

What landed in this repo:

- tests/claude_code/manifest.yaml — feature manifest with one entry
  (basic_messaging_non_streaming) plus the v0 provider column order.
- tests/claude_code/cli_driver.py — Claude Code CLI Driver. One entry
  point (run_claude); handles subprocess assembly, env overlay, stream-JSON
  parsing, and structured failure modes. `runner=` is a unit-test seam.
- tests/claude_code/conftest.py — `compat_result` fixture (tagged-union
  recorder) + pytest_runtest_makereport hook that infers (feature, provider)
  from the file path and writes a structured compat-results.json artifact.
- tests/claude_code/basic_messaging_non_streaming/test_anthropic.py — the
  one cell, parametrized over Haiku/Sonnet/Opus per the PRD's per-cell
  model rule.
- tests/claude_code/matrix_builder.py — pure-function builder from
  (manifest, results, run-metadata) to the v1 JSON schema. Aggregates per-
  model results into one cell (pass iff all pass). build_from_paths is the
  thin I/O wrapper for the publisher.
- tests/claude_code/sample_compatibility-matrix.json — hand-authored sample
  of the v1 JSON; copied to the docs repo by hand as part of this slice.
- Unit tests: 10 driver tests (mocked subprocess), 9 compat_result tests,
  10 matrix-builder golden-file tests. 29/29 pass.

Key decisions:

- (feature, provider) is inferred from file path, not declared in metadata —
  mirrors the PRD's "no drift" goal.
- Driver injects subprocess via a `runner` kwarg so unit tests don't need
  the real `claude` CLI; production callers leave it default.
- Builder is a pure function on Mappings/Sequences; load/write live in a
  thin `build_from_paths` wrapper. Golden-file tests pin the schema.
- `_driver_unit_tests/` and `_builder_unit_tests/` are prefixed with `_`
  so the conftest's path-inference hook skips them and they don't
  pollute the matrix artifact.
- `compat-results.json` added to .gitignore (CI-only output).

Out of scope per CLAUDE.md (docs live in BerriAI/litellm-docs):
- The MDX page `docs/tutorials/claude-code-compatibility` and the
  `<CompatibilityMatrix />` React component. The hand-authored
  compatibility-matrix.json (`sample_compatibility-matrix.json` in this
  repo) is the artifact those docs files will consume; opening that doc
  PR is the next step in this slice.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
mateo-berri 2026-04-25 04:07:02 +00:00
parent 70492cee42
commit 0bf013f620
17 changed files with 1290 additions and 1 deletions

5
.gitignore vendored
View file

@ -100,4 +100,7 @@ STABILIZATION_TODO.md
**/test-results
**/playwright-report
**/*.storageState.json
**/coverage
**/coverage
# Claude Code compatibility-matrix pytest artifact (CI-only output).
compat-results.json

View file

View file

@ -0,0 +1,38 @@
{
"schema_version": "1",
"generated_at": "2026-04-25T00:00:00Z",
"litellm_version": "v1.83.0-stable",
"claude_code_version": "2.1.120",
"providers": [
"anthropic",
"bedrock_invoke"
],
"features": [
{
"id": "basic_messaging_non_streaming",
"name": "Basic messaging (non-streaming)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "not_tested"
}
}
},
{
"id": "tool_use",
"name": "Tool use",
"providers": {
"anthropic": {
"status": "fail",
"error": "[claude-sonnet-4-6] tool call dropped"
},
"bedrock_invoke": {
"status": "not_applicable",
"reason": "tool use not yet wired up for Bedrock Invoke"
}
}
}
]
}

View file

@ -0,0 +1,9 @@
schema_version: "1"
providers:
- anthropic
- bedrock_invoke
features:
- id: basic_messaging_non_streaming
name: Basic messaging (non-streaming)
- id: tool_use
name: Tool use

View file

@ -0,0 +1,41 @@
{
"schema_version": "1",
"results": [
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-haiku-4-5]",
"result": {"status": "pass"}
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-sonnet-4-6]",
"result": {"status": "pass"}
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-opus-4-7]",
"result": {"status": "pass"}
},
{
"feature_id": "tool_use",
"provider": "anthropic",
"nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-haiku-4-5]",
"result": {"status": "pass"}
},
{
"feature_id": "tool_use",
"provider": "anthropic",
"nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-sonnet-4-6]",
"result": {"status": "fail", "error": "[claude-sonnet-4-6] tool call dropped"}
},
{
"feature_id": "tool_use",
"provider": "bedrock_invoke",
"nodeid": "tests/claude_code/tool_use/test_bedrock_invoke.py::test_x[claude-haiku-4-5]",
"result": {"status": "not_applicable", "reason": "tool use not yet wired up for Bedrock Invoke"}
}
]
}

View file

@ -0,0 +1,192 @@
"""Golden-file tests for the Matrix JSON Builder.
These tests fix the published JSON schema. The builder is a pure function
from (manifest, results, metadata) → matrix dict, so we feed it a fixture
input set and compare the produced dict to a checked-in expected output.
Any schema drift — intentional or accidental — surfaces as a diff in PR
review.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from tests.claude_code.matrix_builder import (
ManifestError,
ResultsError,
build_from_paths,
build_matrix,
load_manifest,
load_results,
)
FIXTURES = Path(__file__).parent / "fixtures"
def test_build_matrix_matches_golden_file(tmp_path):
manifest = load_manifest(FIXTURES / "manifest.yaml")
results = load_results(FIXTURES / "results.json")
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v1.83.0-stable",
claude_code_version="2.1.120",
generated_at="2026-04-25T00:00:00Z",
)
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
assert matrix == expected
def test_build_matrix_pass_requires_all_models_pass():
"""Multiple results in one cell must all be pass for the cell to be pass."""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
def test_build_matrix_any_fail_makes_cell_fail():
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "fail", "error": "[claude-opus-4-7] timeout"},
},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cell = matrix["features"][0]["providers"]["anthropic"]
assert cell["status"] == "fail"
assert cell["error"] == "[claude-opus-4-7] timeout"
def test_build_matrix_fills_not_tested_for_missing_cells():
manifest = {
"schema_version": "1",
"providers": ["anthropic", "azure"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cells = matrix["features"][0]["providers"]
assert cells["anthropic"] == {"status": "pass"}
assert cells["azure"] == {"status": "not_tested"}
def test_build_matrix_preserves_provider_and_feature_order():
manifest = {
"schema_version": "1",
"providers": ["azure", "anthropic", "vertex_ai"],
"features": [
{"id": "z", "name": "Z"},
{"id": "a", "name": "A"},
],
}
matrix = build_matrix(
manifest=manifest,
results=[],
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["providers"] == ["azure", "anthropic", "vertex_ai"]
assert [f["id"] for f in matrix["features"]] == ["z", "a"]
assert list(matrix["features"][0]["providers"].keys()) == [
"azure",
"anthropic",
"vertex_ai",
]
def test_build_matrix_emits_schema_version_one():
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
matrix = build_matrix(
manifest=manifest,
results=[],
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["schema_version"] == "1"
def test_load_manifest_rejects_wrong_schema_version(tmp_path):
bad = tmp_path / "manifest.yaml"
bad.write_text(
'schema_version: "2"\nproviders: [anthropic]\nfeatures:\n - id: f\n name: F\n'
)
with pytest.raises(ManifestError, match="schema_version"):
load_manifest(bad)
def test_load_manifest_rejects_empty_features(tmp_path):
bad = tmp_path / "manifest.yaml"
bad.write_text('schema_version: "1"\nproviders: [anthropic]\nfeatures: []\n')
with pytest.raises(ManifestError):
load_manifest(bad)
def test_load_results_rejects_missing_results_key(tmp_path):
bad = tmp_path / "results.json"
bad.write_text(json.dumps({"schema_version": "1"}))
with pytest.raises(ResultsError):
load_results(bad)
def test_build_from_paths_writes_output(tmp_path):
out = tmp_path / "compatibility-matrix.json"
matrix = build_from_paths(
manifest_path=FIXTURES / "manifest.yaml",
results_path=FIXTURES / "results.json",
litellm_version="v1.83.0-stable",
claude_code_version="2.1.120",
generated_at="2026-04-25T00:00:00Z",
output_path=out,
)
assert out.exists()
on_disk = json.loads(out.read_text())
assert on_disk == matrix
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
assert on_disk == expected

View file

@ -0,0 +1,239 @@
"""Unit tests for the Claude Code CLI Driver.
These tests mock the subprocess so they run anywhere — no network, no
`claude` install, no API keys. They cover the behavior contract:
argument assembly, environment overlay, stream-JSON parsing, exit-code
plumbing, and the structured failure modes (CLI not found, timeout).
"""
from __future__ import annotations
import json
import subprocess
from dataclasses import dataclass
from typing import List, Optional
import pytest
from tests.claude_code.cli_driver import (
ClaudeCLIError,
DriverResult,
run_claude,
)
@dataclass
class _Completed:
returncode: int = 0
stdout: str = ""
stderr: str = ""
def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
captured = {}
def runner(cmd, env, capture_output, text, timeout, check):
captured["cmd"] = cmd
captured["env"] = env
captured["timeout"] = timeout
return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
return runner, captured
def test_run_claude_assembles_command_correctly():
runner, captured = _make_runner(
stdout='{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}\n'
)
run_claude(
prompt="hello",
model="claude-haiku-4-5",
base_url="http://localhost:4000",
api_key="sk-test",
runner=runner,
)
cmd = captured["cmd"]
assert cmd[0] == "claude"
assert "--print" in cmd
assert "--output-format" in cmd
assert "stream-json" in cmd
assert "--model" in cmd
assert "claude-haiku-4-5" in cmd
assert cmd[-1] == "hello"
def test_run_claude_overlays_proxy_env():
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://proxy.example:4000",
api_key="sk-abc",
runner=runner,
)
env = captured["env"]
assert env["ANTHROPIC_BASE_URL"] == "http://proxy.example:4000"
assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-abc"
def test_run_claude_extra_env_takes_precedence_over_os_environ(monkeypatch):
monkeypatch.setenv("FOO", "from-os")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
extra_env={"FOO": "from-arg"},
runner=runner,
)
assert captured["env"]["FOO"] == "from-arg"
def test_run_claude_parses_stream_json_assistant_text():
events = [
{"type": "system", "session_id": "abc"},
{
"type": "assistant",
"message": {
"content": [
{"type": "text", "text": "Hello "},
{"type": "text", "text": "world"},
]
},
},
{"type": "result", "usage": {"input_tokens": 10, "output_tokens": 2}},
]
stdout = "\n".join(json.dumps(e) for e in events) + "\n"
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="claude-haiku-4-5",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
assert isinstance(result, DriverResult)
assert result.text == "Hello world"
assert len(result.events) == 3
assert result.usage == {"input_tokens": 10, "output_tokens": 2}
assert result.exit_code == 0
def test_run_claude_handles_string_message_content():
"""Some CLI versions emit `message.content` as a plain string."""
stdout = (
json.dumps({"type": "assistant", "message": {"content": "bare text"}}) + "\n"
)
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.text == "bare text"
def test_run_claude_skips_malformed_lines():
stdout = (
"not-json\n"
+ json.dumps(
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "x"}]},
}
)
+ "\n"
+ "{also-bad\n"
)
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.text == "x"
assert len(result.events) == 1
def test_run_claude_propagates_nonzero_exit_code():
runner, _ = _make_runner(stdout="", returncode=2, stderr="auth failed")
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.exit_code == 2
assert result.stderr == "auth failed"
assert result.text == ""
def test_run_claude_raises_on_missing_cli():
def runner(*args, **kwargs):
raise FileNotFoundError(2, "no such file", "claude")
with pytest.raises(ClaudeCLIError, match="claude CLI not found"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
def test_run_claude_raises_on_timeout():
def runner(*args, **kwargs):
raise subprocess.TimeoutExpired(cmd="claude", timeout=1)
with pytest.raises(ClaudeCLIError, match="timed out"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
timeout=1,
runner=runner,
)
def test_run_claude_validates_required_params():
runner, _ = _make_runner()
with pytest.raises(ValueError, match="prompt"):
run_claude(
prompt="",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="model"):
run_claude(
prompt="hi",
model="",
base_url="http://x",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="base_url"):
run_claude(
prompt="hi",
model="m",
base_url="",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="api_key"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="",
runner=runner,
)

View file

@ -0,0 +1,70 @@
"""Tests for the `compat_result` fixture's tagged-union validation.
The conftest's `pytest_runtest_makereport` hook is exercised end-to-end by
the matrix-builder golden-file tests (which consume a results.json that
the harness would produce). Here we just test the input-validation
contract on `CompatResult.set()`.
"""
from __future__ import annotations
import pytest
from tests.claude_code.conftest import CompatResult
def test_set_pass_is_accepted():
r = CompatResult()
r.set({"status": "pass"})
assert r.value == {"status": "pass"}
def test_set_fail_requires_error():
r = CompatResult()
with pytest.raises(ValueError, match="requires 'error'"):
r.set({"status": "fail"})
def test_set_fail_with_error_is_accepted():
r = CompatResult()
r.set({"status": "fail", "error": "boom"})
assert r.value == {"status": "fail", "error": "boom"}
def test_set_not_applicable_requires_reason():
r = CompatResult()
with pytest.raises(ValueError, match="requires 'reason'"):
r.set({"status": "not_applicable"})
def test_set_not_applicable_with_reason_is_accepted():
r = CompatResult()
r.set({"status": "not_applicable", "reason": "Bedrock has no /thinking"})
assert r.value == {"status": "not_applicable", "reason": "Bedrock has no /thinking"}
def test_set_not_tested_is_accepted():
r = CompatResult()
r.set({"status": "not_tested"})
assert r.value == {"status": "not_tested"}
def test_set_rejects_unknown_status():
r = CompatResult()
with pytest.raises(ValueError, match="status must be one of"):
r.set({"status": "maybe"})
def test_set_rejects_non_dict():
r = CompatResult()
with pytest.raises(TypeError):
r.set("pass") # type: ignore[arg-type]
def test_set_copies_input():
"""Mutating the dict after set() must not change the stored value."""
r = CompatResult()
payload = {"status": "fail", "error": "x"}
r.set(payload)
payload["error"] = "mutated"
assert r.value["error"] == "x"

View file

@ -0,0 +1,98 @@
"""basic_messaging_non_streaming × Anthropic.
The thinnest end-to-end path through every layer of the matrix: drive the
real `claude` CLI in headless mode against a running LiteLLM proxy that
routes to Anthropic, and report the outcome via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. We parametrize over the
three models and the conftest aggregator produces one cell from the three
results.
"""
from __future__ import annotations
import os
import pytest
from tests.claude_code.cli_driver import ClaudeCLIError, run_claude
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Per the PRD: each cell is exercised against three Claude tiers via the
# Anthropic provider. Aliases are configured in the LiteLLM proxy's
# routing config; the driver only sends the alias.
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
@pytest.mark.parametrize("model", ANTHROPIC_MODELS)
def test_basic_messaging_non_streaming_anthropic(compat_result, model):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply.
"Basic messaging" means: send a single user prompt, receive any
non-empty assistant text reply, no tools, no streaming, no thinking.
The whole point of this slice is to prove the path works at all —
so the assertion is intentionally lenient on the reply contents.
"""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
try:
result = run_claude(
prompt="Reply with the single word 'pong' and nothing else.",
model=model,
base_url=base_url,
api_key=api_key,
)
except ClaudeCLIError as exc:
compat_result.set({"status": "fail", "error": f"[{model}] {exc}"})
pytest.fail(str(exc), pytrace=False)
return
if result.exit_code != 0:
compat_result.set(
{
"status": "fail",
"error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}",
}
)
pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False)
return
if not result.text.strip():
compat_result.set(
{
"status": "fail",
"error": f"[{model}] claude returned empty assistant text",
}
)
pytest.fail(f"empty reply for {model}", pytrace=False)
return
compat_result.set({"status": "pass"})

View file

@ -0,0 +1,194 @@
"""Claude Code CLI Driver.
A thin wrapper around the `claude` CLI in headless mode. Every compatibility
test consumes only this module — tests must never shell out directly. This
keeps the subprocess assembly, stream-JSON parsing, and result shape in a
single place that can be unit-tested with a mocked subprocess.
The driver is deliberately small: it knows how to invoke the CLI, drain its
stream-JSON output, and return a structured `DriverResult`. Higher-level
matrix concerns (status aggregation, manifest lookup, JSON serialization)
live in `matrix_builder.py`.
"""
from __future__ import annotations
import json
import os
import subprocess
from dataclasses import dataclass, field
from typing import Any, Dict, List, Mapping, Optional, Sequence
CLAUDE_CLI_DEFAULT = "claude"
DEFAULT_TIMEOUT_SECONDS = 120
class ClaudeCLIError(RuntimeError):
"""Raised when the `claude` CLI cannot be invoked or returns a fatal error."""
@dataclass
class DriverResult:
"""Structured outcome of a single `claude` CLI invocation.
`text` is the assistant's final user-visible reply (joined across any
intermediate `assistant` events for non-streaming runs). `events` is the
raw list of stream-JSON objects emitted by the CLI, preserved so test
authors can write feature-specific assertions (tool calls, cache hits,
usage) without re-parsing stdout.
"""
text: str
events: List[Dict[str, Any]] = field(default_factory=list)
exit_code: int = 0
stderr: str = ""
usage: Optional[Dict[str, Any]] = None
duration_ms: Optional[int] = None
def run_claude(
*,
prompt: str,
model: str,
base_url: str,
api_key: str,
extra_env: Optional[Mapping[str, str]] = None,
extra_args: Optional[Sequence[str]] = None,
cli_path: str = CLAUDE_CLI_DEFAULT,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
runner: Optional[Any] = None,
) -> DriverResult:
"""Invoke `claude` once in headless stream-JSON mode and return the result.
The CLI is pointed at a LiteLLM proxy via `ANTHROPIC_BASE_URL` /
`ANTHROPIC_AUTH_TOKEN`, so the same code path exercises every provider
column — only the model id and the proxy's routing differ between
invocations.
`runner` is an injection seam used by the unit tests: by default we call
`subprocess.run`, but the test suite swaps in a fake that yields canned
stream-JSON. Production callers should never set it.
"""
if not prompt:
raise ValueError("prompt must be a non-empty string")
if not model:
raise ValueError("model must be a non-empty string")
if not base_url:
raise ValueError("base_url must be a non-empty string")
if not api_key:
raise ValueError("api_key must be a non-empty string")
cmd: List[str] = [
cli_path,
"--print",
"--output-format",
"stream-json",
"--verbose",
"--model",
model,
prompt,
]
if extra_args:
cmd.extend(extra_args)
env = {**os.environ, **(extra_env or {})}
env["ANTHROPIC_BASE_URL"] = base_url
env["ANTHROPIC_AUTH_TOKEN"] = api_key
run_fn = runner or subprocess.run
try:
completed = run_fn(
cmd,
env=env,
capture_output=True,
text=True,
timeout=timeout,
check=False,
)
except FileNotFoundError as exc:
raise ClaudeCLIError(
f"claude CLI not found at {cli_path!r}; install with `npm i -g @anthropic-ai/claude-code`"
) from exc
except subprocess.TimeoutExpired as exc:
raise ClaudeCLIError(f"claude CLI timed out after {timeout}s") from exc
events = _parse_stream_json(completed.stdout or "")
text = _extract_assistant_text(events)
usage = _extract_usage(events)
return DriverResult(
text=text,
events=events,
exit_code=completed.returncode,
stderr=completed.stderr or "",
usage=usage,
)
def _parse_stream_json(stdout: str) -> List[Dict[str, Any]]:
"""Parse newline-delimited JSON emitted by `claude --output-format stream-json`.
Lines that don't parse as JSON are silently skipped — the CLI occasionally
emits debug output we don't care about, and a single malformed line should
not abort the whole run. Real failure modes surface via exit code.
"""
events: List[Dict[str, Any]] = []
for line in stdout.splitlines():
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(obj, dict):
events.append(obj)
return events
def _extract_assistant_text(events: Sequence[Mapping[str, Any]]) -> str:
"""Concatenate the text content of every `assistant` event in order.
The non-streaming `--print` path emits a single `assistant` event whose
`message.content` is a list of content blocks. We walk the blocks and
join every `text` block — the CLI prints other block types (e.g.
`tool_use`) which we ignore for the basic-messaging case.
"""
chunks: List[str] = []
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if isinstance(content, str):
chunks.append(content)
continue
if not isinstance(content, list):
continue
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") == "text" and isinstance(block.get("text"), str):
chunks.append(block["text"])
return "".join(chunks)
def _extract_usage(events: Sequence[Mapping[str, Any]]) -> Optional[Dict[str, Any]]:
"""Return the most recent `usage` block seen on any event, if any.
The CLI surfaces token + cache usage on the final `result` event for
non-streaming runs, but earlier events also carry partial usage in some
versions; taking the last non-empty one is the safe default.
"""
last: Optional[Dict[str, Any]] = None
for event in events:
usage = event.get("usage")
if isinstance(usage, dict) and usage:
last = usage
continue
message = event.get("message")
if isinstance(message, dict):
inner = message.get("usage")
if isinstance(inner, dict) and inner:
last = inner
return last

View file

@ -0,0 +1,164 @@
"""Pytest plumbing for the Claude Code compatibility matrix.
Two responsibilities live here:
1. The `compat_result` fixture — the only API a test author needs to learn.
Tests call `compat_result.set({"status": "pass"})` (or fail / not_applicable)
to report their outcome as a tagged union. The fixture is per-test and
stores the last value reported.
2. The `pytest_runtest_makereport` hook — captures each test's reported result,
infers (feature, provider) from the file path, and writes a single
`compat-results.json` artifact next to JUnit XML. The Matrix JSON Builder
consumes this artifact to produce the published `compatibility-matrix.json`.
The (feature, provider) inference comes from the test file path: the parent
directory name is the feature_id (matching `manifest.yaml`), and the file
stem after the leading `test_` is the provider id. This avoids per-file
metadata that drifts.
"""
from __future__ import annotations
import json
import os
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Optional
import pytest
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
RESULTS_ARTIFACT_ENV = "COMPAT_RESULTS_PATH"
DEFAULT_ARTIFACT_PATH = "compat-results.json"
@dataclass
class CompatResult:
"""Per-test recorder for compatibility outcomes.
Tests interact only via `.set(...)`. `.value` is read by the
`pytest_runtest_makereport` hook after the test body finishes.
"""
value: Optional[Dict[str, Any]] = None
def set(self, result: Dict[str, Any]) -> None:
if not isinstance(result, dict):
raise TypeError("compat_result.set() requires a dict")
status = result.get("status")
if status not in VALID_STATUSES:
raise ValueError(
f"compat_result.set() status must be one of {sorted(VALID_STATUSES)}, "
f"got {status!r}"
)
if status == "fail" and not result.get("error"):
raise ValueError("compat_result.set({'status': 'fail'}) requires 'error'")
if status == "not_applicable" and not result.get("reason"):
raise ValueError(
"compat_result.set({'status': 'not_applicable'}) requires 'reason'"
)
self.value = dict(result)
@dataclass
class _CollectedResult:
feature_id: str
provider: str
nodeid: str
result: Dict[str, Any]
@dataclass
class _Collector:
items: List[_CollectedResult] = field(default_factory=list)
_COLLECTOR = _Collector()
@pytest.fixture
def compat_result() -> CompatResult:
"""Per-test recorder for the (feature, provider) outcome.
Tests should call `compat_result.set({"status": "pass"})` (or fail /
not_applicable) before returning. If a test exits without calling `.set()`
the harness records `status="fail"` with an explanatory error so that
every collected node maps to a real cell.
"""
return CompatResult()
def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]:
"""Infer (feature_id, provider) from a test file path.
Path shape: tests/claude_code/<feature_id>/test_<provider>.py
Returns None if the file is not a per-feature test (e.g. unit tests
living under tests/claude_code/_driver_unit_tests/), so those don't
pollute the matrix artifact.
"""
name = node_path.name
if not name.startswith("test_") or not name.endswith(".py"):
return None
provider = name[len("test_") : -len(".py")]
feature_id = node_path.parent.name
if feature_id.startswith("_") or feature_id == "claude_code":
return None
return feature_id, provider
@pytest.hookimpl(hookwrapper=True)
def pytest_runtest_makereport(item, call):
"""Capture compat_result.value at end-of-test and remember it for the artifact."""
outcome = yield
report = outcome.get_result()
if report.when != "call":
return
inferred = _infer_feature_and_provider(Path(str(item.path)))
if inferred is None:
return
feature_id, provider = inferred
fixture = item.funcargs.get("compat_result") if hasattr(item, "funcargs") else None
reported: Optional[Dict[str, Any]] = getattr(fixture, "value", None)
if reported is None:
if report.passed:
reported = {
"status": "fail",
"error": "test passed without calling compat_result.set(); "
"every compat test must report a status.",
}
else:
reported = {
"status": "fail",
"error": (str(report.longrepr) if report.longrepr else "test failed"),
}
_COLLECTOR.items.append(
_CollectedResult(
feature_id=feature_id,
provider=provider,
nodeid=report.nodeid,
result=reported,
)
)
def pytest_sessionfinish(session, exitstatus):
"""Write the structured results artifact at end of session."""
artifact_path = os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH
payload = {
"schema_version": "1",
"results": [
{
"feature_id": item.feature_id,
"provider": item.provider,
"nodeid": item.nodeid,
"result": item.result,
}
for item in _COLLECTOR.items
],
}
Path(artifact_path).write_text(json.dumps(payload, indent=2, sort_keys=True))

View file

@ -0,0 +1,26 @@
# Claude Code Compatibility Matrix — feature manifest.
#
# Defines the row order of the matrix and maps each feature_id to its
# human-readable display name. Adding a new feature to the matrix is a
# three-step change:
# 1. Append an entry to `features:` below.
# 2. Create a directory `tests/claude_code/<feature_id>/`.
# 3. Add per-provider test files inside that directory.
#
# `feature_id` MUST match the directory name on disk; the test harness
# infers (feature, provider) for each test from its file path.
schema_version: "1"
# Provider column order in the rendered matrix.
providers:
- anthropic
- bedrock_invoke
- bedrock_converse
- vertex_ai
- azure
# Feature row order.
features:
- id: basic_messaging_non_streaming
name: Basic messaging (non-streaming)

View file

@ -0,0 +1,179 @@
"""Matrix JSON Builder.
Pure-function module that consumes the pytest-produced `compat-results.json`,
the manifest, and run metadata, and emits the final `compatibility-matrix.json`
conforming to the schema published in the PRD.
This module is deliberately free of subprocess, network, or filesystem side
effects in its public API — the public entry points take pre-loaded inputs
and return data structures, so they can be exercised by golden-file tests
without I/O. A small `build_from_paths()` convenience wrapper does the I/O
for callers that need it (the daily-cron publisher).
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Dict, List, Mapping, Optional, Sequence
import yaml
SCHEMA_VERSION = "1"
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
class ManifestError(ValueError):
"""Raised when `manifest.yaml` is malformed."""
class ResultsError(ValueError):
"""Raised when the pytest results artifact is malformed."""
def load_manifest(path: Path) -> Dict[str, Any]:
"""Load and validate `manifest.yaml`.
Returns a dict with keys: schema_version, providers, features. Raises
ManifestError on missing fields or schema mismatch.
"""
raw = yaml.safe_load(path.read_text())
if not isinstance(raw, dict):
raise ManifestError(f"manifest at {path} is not a mapping")
schema_version = str(raw.get("schema_version", ""))
if schema_version != SCHEMA_VERSION:
raise ManifestError(
f"manifest schema_version {schema_version!r} does not match "
f"builder version {SCHEMA_VERSION!r}"
)
providers = raw.get("providers")
if not isinstance(providers, list) or not providers:
raise ManifestError("manifest.providers must be a non-empty list")
features = raw.get("features")
if not isinstance(features, list) or not features:
raise ManifestError("manifest.features must be a non-empty list")
for feature in features:
if not isinstance(feature, dict):
raise ManifestError("each feature must be a mapping")
if not feature.get("id") or not feature.get("name"):
raise ManifestError("each feature must have id and name")
return raw
def load_results(path: Path) -> List[Dict[str, Any]]:
"""Load the pytest results artifact and return its `results` list."""
raw = json.loads(path.read_text())
if not isinstance(raw, dict) or not isinstance(raw.get("results"), list):
raise ResultsError(f"results artifact at {path} has no `results` list")
return raw["results"]
def build_matrix(
*,
manifest: Mapping[str, Any],
results: Sequence[Mapping[str, Any]],
litellm_version: str,
claude_code_version: str,
generated_at: str,
) -> Dict[str, Any]:
"""Build the published matrix JSON from pre-loaded inputs.
Empty cells (no test ran for a (feature, provider) and no
`not_applicable` was declared) are filled in with `not_tested`. If
multiple results report on the same cell — e.g. a per-feature test
file containing one parametrize per Claude model — the cell aggregates
to `pass` only if every model passed; otherwise `fail` with the first
breaking model surfaced in the error.
"""
providers: List[str] = list(manifest["providers"])
feature_specs: List[Dict[str, Any]] = list(manifest["features"])
grouped: Dict[tuple, List[Dict[str, Any]]] = {}
for entry in results:
if not isinstance(entry, Mapping):
continue
feature_id = entry.get("feature_id")
provider = entry.get("provider")
result = entry.get("result")
if not feature_id or not provider or not isinstance(result, Mapping):
continue
if result.get("status") not in VALID_STATUSES:
continue
grouped.setdefault((feature_id, provider), []).append(dict(result))
features_out: List[Dict[str, Any]] = []
for spec in feature_specs:
feature_id = spec["id"]
cells: Dict[str, Dict[str, Any]] = {}
for provider in providers:
cell_results = grouped.get((feature_id, provider), [])
cells[provider] = _aggregate_cell(cell_results)
features_out.append(
{
"id": feature_id,
"name": spec["name"],
"providers": cells,
}
)
return {
"schema_version": SCHEMA_VERSION,
"generated_at": generated_at,
"litellm_version": litellm_version,
"claude_code_version": claude_code_version,
"providers": providers,
"features": features_out,
}
def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
"""Aggregate a list of per-model results into a single cell status.
Order of precedence (most informative wins):
- Any `fail` → cell is `fail` with the first failure's error.
- `not_applicable` → cell is `not_applicable` with the reason.
- `pass` → cell is `pass`.
- empty / nothing recognized → `not_tested`.
"""
if not results:
return {"status": "not_tested"}
for r in results:
if r.get("status") == "fail":
return {"status": "fail", "error": str(r.get("error", "test failed"))}
for r in results:
if r.get("status") == "not_applicable":
return {
"status": "not_applicable",
"reason": str(r.get("reason", "not applicable")),
}
if all(r.get("status") == "pass" for r in results):
return {"status": "pass"}
return {"status": "not_tested"}
def build_from_paths(
*,
manifest_path: Path,
results_path: Path,
litellm_version: str,
claude_code_version: str,
generated_at: str,
output_path: Optional[Path] = None,
) -> Dict[str, Any]:
"""I/O wrapper around build_matrix used by the publisher script."""
manifest = load_manifest(manifest_path)
results = load_results(results_path)
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version=litellm_version,
claude_code_version=claude_code_version,
generated_at=generated_at,
)
if output_path is not None:
output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n")
return matrix

View file

@ -0,0 +1,36 @@
{
"schema_version": "1",
"generated_at": "2026-04-25T00:00:00Z",
"litellm_version": "v1.83.0-stable",
"claude_code_version": "2.1.120",
"providers": [
"anthropic",
"bedrock_invoke",
"bedrock_converse",
"vertex_ai",
"azure"
],
"features": [
{
"id": "basic_messaging_non_streaming",
"name": "Basic messaging (non-streaming)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "not_tested"
},
"bedrock_converse": {
"status": "not_tested"
},
"vertex_ai": {
"status": "not_tested"
},
"azure": {
"status": "not_tested"
}
}
}
]
}