mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Slice 1 of the Claude Code Compatibility Matrix: the thinnest end-to-end path through every layer for a single (feature, provider) cell, so a future docs page can render a real green cell sourced from a real test. What landed in this repo: - tests/claude_code/manifest.yaml — feature manifest with one entry (basic_messaging_non_streaming) plus the v0 provider column order. - tests/claude_code/cli_driver.py — Claude Code CLI Driver. One entry point (run_claude); handles subprocess assembly, env overlay, stream-JSON parsing, and structured failure modes. `runner=` is a unit-test seam. - tests/claude_code/conftest.py — `compat_result` fixture (tagged-union recorder) + pytest_runtest_makereport hook that infers (feature, provider) from the file path and writes a structured compat-results.json artifact. - tests/claude_code/basic_messaging_non_streaming/test_anthropic.py — the one cell, parametrized over Haiku/Sonnet/Opus per the PRD's per-cell model rule. - tests/claude_code/matrix_builder.py — pure-function builder from (manifest, results, run-metadata) to the v1 JSON schema. Aggregates per- model results into one cell (pass iff all pass). build_from_paths is the thin I/O wrapper for the publisher. - tests/claude_code/sample_compatibility-matrix.json — hand-authored sample of the v1 JSON; copied to the docs repo by hand as part of this slice. - Unit tests: 10 driver tests (mocked subprocess), 9 compat_result tests, 10 matrix-builder golden-file tests. 29/29 pass. Key decisions: - (feature, provider) is inferred from file path, not declared in metadata — mirrors the PRD's "no drift" goal. - Driver injects subprocess via a `runner` kwarg so unit tests don't need the real `claude` CLI; production callers leave it default. - Builder is a pure function on Mappings/Sequences; load/write live in a thin `build_from_paths` wrapper. Golden-file tests pin the schema. - `_driver_unit_tests/` and `_builder_unit_tests/` are prefixed with `_` so the conftest's path-inference hook skips them and they don't pollute the matrix artifact. - `compat-results.json` added to .gitignore (CI-only output). Out of scope per CLAUDE.md (docs live in BerriAI/litellm-docs): - The MDX page `docs/tutorials/claude-code-compatibility` and the `<CompatibilityMatrix />` React component. The hand-authored compatibility-matrix.json (`sample_compatibility-matrix.json` in this repo) is the artifact those docs files will consume; opening that doc PR is the next step in this slice. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
parent
70492cee42
commit
0bf013f620
17 changed files with 1290 additions and 1 deletions
5
.gitignore
vendored
5
.gitignore
vendored
|
|
@ -100,4 +100,7 @@ STABILIZATION_TODO.md
|
|||
**/test-results
|
||||
**/playwright-report
|
||||
**/*.storageState.json
|
||||
**/coverage
|
||||
**/coverage
|
||||
|
||||
# Claude Code compatibility-matrix pytest artifact (CI-only output).
|
||||
compat-results.json
|
||||
0
tests/claude_code/__init__.py
Normal file
0
tests/claude_code/__init__.py
Normal file
0
tests/claude_code/_builder_unit_tests/__init__.py
Normal file
0
tests/claude_code/_builder_unit_tests/__init__.py
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
{
|
||||
"schema_version": "1",
|
||||
"generated_at": "2026-04-25T00:00:00Z",
|
||||
"litellm_version": "v1.83.0-stable",
|
||||
"claude_code_version": "2.1.120",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"bedrock_invoke"
|
||||
],
|
||||
"features": [
|
||||
{
|
||||
"id": "basic_messaging_non_streaming",
|
||||
"name": "Basic messaging (non-streaming)",
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"status": "pass"
|
||||
},
|
||||
"bedrock_invoke": {
|
||||
"status": "not_tested"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "tool_use",
|
||||
"name": "Tool use",
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"status": "fail",
|
||||
"error": "[claude-sonnet-4-6] tool call dropped"
|
||||
},
|
||||
"bedrock_invoke": {
|
||||
"status": "not_applicable",
|
||||
"reason": "tool use not yet wired up for Bedrock Invoke"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
schema_version: "1"
|
||||
providers:
|
||||
- anthropic
|
||||
- bedrock_invoke
|
||||
features:
|
||||
- id: basic_messaging_non_streaming
|
||||
name: Basic messaging (non-streaming)
|
||||
- id: tool_use
|
||||
name: Tool use
|
||||
41
tests/claude_code/_builder_unit_tests/fixtures/results.json
Normal file
41
tests/claude_code/_builder_unit_tests/fixtures/results.json
Normal file
|
|
@ -0,0 +1,41 @@
|
|||
{
|
||||
"schema_version": "1",
|
||||
"results": [
|
||||
{
|
||||
"feature_id": "basic_messaging_non_streaming",
|
||||
"provider": "anthropic",
|
||||
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-haiku-4-5]",
|
||||
"result": {"status": "pass"}
|
||||
},
|
||||
{
|
||||
"feature_id": "basic_messaging_non_streaming",
|
||||
"provider": "anthropic",
|
||||
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-sonnet-4-6]",
|
||||
"result": {"status": "pass"}
|
||||
},
|
||||
{
|
||||
"feature_id": "basic_messaging_non_streaming",
|
||||
"provider": "anthropic",
|
||||
"nodeid": "tests/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-opus-4-7]",
|
||||
"result": {"status": "pass"}
|
||||
},
|
||||
{
|
||||
"feature_id": "tool_use",
|
||||
"provider": "anthropic",
|
||||
"nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-haiku-4-5]",
|
||||
"result": {"status": "pass"}
|
||||
},
|
||||
{
|
||||
"feature_id": "tool_use",
|
||||
"provider": "anthropic",
|
||||
"nodeid": "tests/claude_code/tool_use/test_anthropic.py::test_x[claude-sonnet-4-6]",
|
||||
"result": {"status": "fail", "error": "[claude-sonnet-4-6] tool call dropped"}
|
||||
},
|
||||
{
|
||||
"feature_id": "tool_use",
|
||||
"provider": "bedrock_invoke",
|
||||
"nodeid": "tests/claude_code/tool_use/test_bedrock_invoke.py::test_x[claude-haiku-4-5]",
|
||||
"result": {"status": "not_applicable", "reason": "tool use not yet wired up for Bedrock Invoke"}
|
||||
}
|
||||
]
|
||||
}
|
||||
192
tests/claude_code/_builder_unit_tests/test_matrix_builder.py
Normal file
192
tests/claude_code/_builder_unit_tests/test_matrix_builder.py
Normal file
|
|
@ -0,0 +1,192 @@
|
|||
"""Golden-file tests for the Matrix JSON Builder.
|
||||
|
||||
These tests fix the published JSON schema. The builder is a pure function
|
||||
from (manifest, results, metadata) → matrix dict, so we feed it a fixture
|
||||
input set and compare the produced dict to a checked-in expected output.
|
||||
|
||||
Any schema drift — intentional or accidental — surfaces as a diff in PR
|
||||
review.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.matrix_builder import (
|
||||
ManifestError,
|
||||
ResultsError,
|
||||
build_from_paths,
|
||||
build_matrix,
|
||||
load_manifest,
|
||||
load_results,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def test_build_matrix_matches_golden_file(tmp_path):
|
||||
manifest = load_manifest(FIXTURES / "manifest.yaml")
|
||||
results = load_results(FIXTURES / "results.json")
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=results,
|
||||
litellm_version="v1.83.0-stable",
|
||||
claude_code_version="2.1.120",
|
||||
generated_at="2026-04-25T00:00:00Z",
|
||||
)
|
||||
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
|
||||
assert matrix == expected
|
||||
|
||||
|
||||
def test_build_matrix_pass_requires_all_models_pass():
|
||||
"""Multiple results in one cell must all be pass for the cell to be pass."""
|
||||
manifest = {
|
||||
"schema_version": "1",
|
||||
"providers": ["anthropic"],
|
||||
"features": [{"id": "f", "name": "F"}],
|
||||
}
|
||||
results = [
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
]
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=results,
|
||||
litellm_version="v",
|
||||
claude_code_version="c",
|
||||
generated_at="t",
|
||||
)
|
||||
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
|
||||
|
||||
|
||||
def test_build_matrix_any_fail_makes_cell_fail():
|
||||
manifest = {
|
||||
"schema_version": "1",
|
||||
"providers": ["anthropic"],
|
||||
"features": [{"id": "f", "name": "F"}],
|
||||
}
|
||||
results = [
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
{
|
||||
"feature_id": "f",
|
||||
"provider": "anthropic",
|
||||
"result": {"status": "fail", "error": "[claude-opus-4-7] timeout"},
|
||||
},
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
]
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=results,
|
||||
litellm_version="v",
|
||||
claude_code_version="c",
|
||||
generated_at="t",
|
||||
)
|
||||
cell = matrix["features"][0]["providers"]["anthropic"]
|
||||
assert cell["status"] == "fail"
|
||||
assert cell["error"] == "[claude-opus-4-7] timeout"
|
||||
|
||||
|
||||
def test_build_matrix_fills_not_tested_for_missing_cells():
|
||||
manifest = {
|
||||
"schema_version": "1",
|
||||
"providers": ["anthropic", "azure"],
|
||||
"features": [{"id": "f", "name": "F"}],
|
||||
}
|
||||
results = [
|
||||
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
|
||||
]
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=results,
|
||||
litellm_version="v",
|
||||
claude_code_version="c",
|
||||
generated_at="t",
|
||||
)
|
||||
cells = matrix["features"][0]["providers"]
|
||||
assert cells["anthropic"] == {"status": "pass"}
|
||||
assert cells["azure"] == {"status": "not_tested"}
|
||||
|
||||
|
||||
def test_build_matrix_preserves_provider_and_feature_order():
|
||||
manifest = {
|
||||
"schema_version": "1",
|
||||
"providers": ["azure", "anthropic", "vertex_ai"],
|
||||
"features": [
|
||||
{"id": "z", "name": "Z"},
|
||||
{"id": "a", "name": "A"},
|
||||
],
|
||||
}
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=[],
|
||||
litellm_version="v",
|
||||
claude_code_version="c",
|
||||
generated_at="t",
|
||||
)
|
||||
assert matrix["providers"] == ["azure", "anthropic", "vertex_ai"]
|
||||
assert [f["id"] for f in matrix["features"]] == ["z", "a"]
|
||||
assert list(matrix["features"][0]["providers"].keys()) == [
|
||||
"azure",
|
||||
"anthropic",
|
||||
"vertex_ai",
|
||||
]
|
||||
|
||||
|
||||
def test_build_matrix_emits_schema_version_one():
|
||||
manifest = {
|
||||
"schema_version": "1",
|
||||
"providers": ["anthropic"],
|
||||
"features": [{"id": "f", "name": "F"}],
|
||||
}
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=[],
|
||||
litellm_version="v",
|
||||
claude_code_version="c",
|
||||
generated_at="t",
|
||||
)
|
||||
assert matrix["schema_version"] == "1"
|
||||
|
||||
|
||||
def test_load_manifest_rejects_wrong_schema_version(tmp_path):
|
||||
bad = tmp_path / "manifest.yaml"
|
||||
bad.write_text(
|
||||
'schema_version: "2"\nproviders: [anthropic]\nfeatures:\n - id: f\n name: F\n'
|
||||
)
|
||||
with pytest.raises(ManifestError, match="schema_version"):
|
||||
load_manifest(bad)
|
||||
|
||||
|
||||
def test_load_manifest_rejects_empty_features(tmp_path):
|
||||
bad = tmp_path / "manifest.yaml"
|
||||
bad.write_text('schema_version: "1"\nproviders: [anthropic]\nfeatures: []\n')
|
||||
with pytest.raises(ManifestError):
|
||||
load_manifest(bad)
|
||||
|
||||
|
||||
def test_load_results_rejects_missing_results_key(tmp_path):
|
||||
bad = tmp_path / "results.json"
|
||||
bad.write_text(json.dumps({"schema_version": "1"}))
|
||||
with pytest.raises(ResultsError):
|
||||
load_results(bad)
|
||||
|
||||
|
||||
def test_build_from_paths_writes_output(tmp_path):
|
||||
out = tmp_path / "compatibility-matrix.json"
|
||||
matrix = build_from_paths(
|
||||
manifest_path=FIXTURES / "manifest.yaml",
|
||||
results_path=FIXTURES / "results.json",
|
||||
litellm_version="v1.83.0-stable",
|
||||
claude_code_version="2.1.120",
|
||||
generated_at="2026-04-25T00:00:00Z",
|
||||
output_path=out,
|
||||
)
|
||||
assert out.exists()
|
||||
on_disk = json.loads(out.read_text())
|
||||
assert on_disk == matrix
|
||||
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
|
||||
assert on_disk == expected
|
||||
0
tests/claude_code/_driver_unit_tests/__init__.py
Normal file
0
tests/claude_code/_driver_unit_tests/__init__.py
Normal file
239
tests/claude_code/_driver_unit_tests/test_cli_driver.py
Normal file
239
tests/claude_code/_driver_unit_tests/test_cli_driver.py
Normal file
|
|
@ -0,0 +1,239 @@
|
|||
"""Unit tests for the Claude Code CLI Driver.
|
||||
|
||||
These tests mock the subprocess so they run anywhere — no network, no
|
||||
`claude` install, no API keys. They cover the behavior contract:
|
||||
argument assembly, environment overlay, stream-JSON parsing, exit-code
|
||||
plumbing, and the structured failure modes (CLI not found, timeout).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Optional
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
DriverResult,
|
||||
run_claude,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class _Completed:
|
||||
returncode: int = 0
|
||||
stdout: str = ""
|
||||
stderr: str = ""
|
||||
|
||||
|
||||
def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
|
||||
captured = {}
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
captured["cmd"] = cmd
|
||||
captured["env"] = env
|
||||
captured["timeout"] = timeout
|
||||
return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
|
||||
|
||||
return runner, captured
|
||||
|
||||
|
||||
def test_run_claude_assembles_command_correctly():
|
||||
runner, captured = _make_runner(
|
||||
stdout='{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}\n'
|
||||
)
|
||||
run_claude(
|
||||
prompt="hello",
|
||||
model="claude-haiku-4-5",
|
||||
base_url="http://localhost:4000",
|
||||
api_key="sk-test",
|
||||
runner=runner,
|
||||
)
|
||||
cmd = captured["cmd"]
|
||||
assert cmd[0] == "claude"
|
||||
assert "--print" in cmd
|
||||
assert "--output-format" in cmd
|
||||
assert "stream-json" in cmd
|
||||
assert "--model" in cmd
|
||||
assert "claude-haiku-4-5" in cmd
|
||||
assert cmd[-1] == "hello"
|
||||
|
||||
|
||||
def test_run_claude_overlays_proxy_env():
|
||||
runner, captured = _make_runner(stdout="")
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="claude-opus-4-7",
|
||||
base_url="http://proxy.example:4000",
|
||||
api_key="sk-abc",
|
||||
runner=runner,
|
||||
)
|
||||
env = captured["env"]
|
||||
assert env["ANTHROPIC_BASE_URL"] == "http://proxy.example:4000"
|
||||
assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-abc"
|
||||
|
||||
|
||||
def test_run_claude_extra_env_takes_precedence_over_os_environ(monkeypatch):
|
||||
monkeypatch.setenv("FOO", "from-os")
|
||||
runner, captured = _make_runner(stdout="")
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="claude-opus-4-7",
|
||||
base_url="http://localhost",
|
||||
api_key="sk-abc",
|
||||
extra_env={"FOO": "from-arg"},
|
||||
runner=runner,
|
||||
)
|
||||
assert captured["env"]["FOO"] == "from-arg"
|
||||
|
||||
|
||||
def test_run_claude_parses_stream_json_assistant_text():
|
||||
events = [
|
||||
{"type": "system", "session_id": "abc"},
|
||||
{
|
||||
"type": "assistant",
|
||||
"message": {
|
||||
"content": [
|
||||
{"type": "text", "text": "Hello "},
|
||||
{"type": "text", "text": "world"},
|
||||
]
|
||||
},
|
||||
},
|
||||
{"type": "result", "usage": {"input_tokens": 10, "output_tokens": 2}},
|
||||
]
|
||||
stdout = "\n".join(json.dumps(e) for e in events) + "\n"
|
||||
runner, _ = _make_runner(stdout=stdout)
|
||||
result = run_claude(
|
||||
prompt="hi",
|
||||
model="claude-haiku-4-5",
|
||||
base_url="http://localhost",
|
||||
api_key="sk-abc",
|
||||
runner=runner,
|
||||
)
|
||||
assert isinstance(result, DriverResult)
|
||||
assert result.text == "Hello world"
|
||||
assert len(result.events) == 3
|
||||
assert result.usage == {"input_tokens": 10, "output_tokens": 2}
|
||||
assert result.exit_code == 0
|
||||
|
||||
|
||||
def test_run_claude_handles_string_message_content():
|
||||
"""Some CLI versions emit `message.content` as a plain string."""
|
||||
stdout = (
|
||||
json.dumps({"type": "assistant", "message": {"content": "bare text"}}) + "\n"
|
||||
)
|
||||
runner, _ = _make_runner(stdout=stdout)
|
||||
result = run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
assert result.text == "bare text"
|
||||
|
||||
|
||||
def test_run_claude_skips_malformed_lines():
|
||||
stdout = (
|
||||
"not-json\n"
|
||||
+ json.dumps(
|
||||
{
|
||||
"type": "assistant",
|
||||
"message": {"content": [{"type": "text", "text": "x"}]},
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
+ "{also-bad\n"
|
||||
)
|
||||
runner, _ = _make_runner(stdout=stdout)
|
||||
result = run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
assert result.text == "x"
|
||||
assert len(result.events) == 1
|
||||
|
||||
|
||||
def test_run_claude_propagates_nonzero_exit_code():
|
||||
runner, _ = _make_runner(stdout="", returncode=2, stderr="auth failed")
|
||||
result = run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
assert result.exit_code == 2
|
||||
assert result.stderr == "auth failed"
|
||||
assert result.text == ""
|
||||
|
||||
|
||||
def test_run_claude_raises_on_missing_cli():
|
||||
def runner(*args, **kwargs):
|
||||
raise FileNotFoundError(2, "no such file", "claude")
|
||||
|
||||
with pytest.raises(ClaudeCLIError, match="claude CLI not found"):
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
|
||||
|
||||
def test_run_claude_raises_on_timeout():
|
||||
def runner(*args, **kwargs):
|
||||
raise subprocess.TimeoutExpired(cmd="claude", timeout=1)
|
||||
|
||||
with pytest.raises(ClaudeCLIError, match="timed out"):
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
timeout=1,
|
||||
runner=runner,
|
||||
)
|
||||
|
||||
|
||||
def test_run_claude_validates_required_params():
|
||||
runner, _ = _make_runner()
|
||||
with pytest.raises(ValueError, match="prompt"):
|
||||
run_claude(
|
||||
prompt="",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
with pytest.raises(ValueError, match="model"):
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="",
|
||||
base_url="http://x",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
with pytest.raises(ValueError, match="base_url"):
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="",
|
||||
api_key="k",
|
||||
runner=runner,
|
||||
)
|
||||
with pytest.raises(ValueError, match="api_key"):
|
||||
run_claude(
|
||||
prompt="hi",
|
||||
model="m",
|
||||
base_url="http://x",
|
||||
api_key="",
|
||||
runner=runner,
|
||||
)
|
||||
70
tests/claude_code/_driver_unit_tests/test_compat_result.py
Normal file
70
tests/claude_code/_driver_unit_tests/test_compat_result.py
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
"""Tests for the `compat_result` fixture's tagged-union validation.
|
||||
|
||||
The conftest's `pytest_runtest_makereport` hook is exercised end-to-end by
|
||||
the matrix-builder golden-file tests (which consume a results.json that
|
||||
the harness would produce). Here we just test the input-validation
|
||||
contract on `CompatResult.set()`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.conftest import CompatResult
|
||||
|
||||
|
||||
def test_set_pass_is_accepted():
|
||||
r = CompatResult()
|
||||
r.set({"status": "pass"})
|
||||
assert r.value == {"status": "pass"}
|
||||
|
||||
|
||||
def test_set_fail_requires_error():
|
||||
r = CompatResult()
|
||||
with pytest.raises(ValueError, match="requires 'error'"):
|
||||
r.set({"status": "fail"})
|
||||
|
||||
|
||||
def test_set_fail_with_error_is_accepted():
|
||||
r = CompatResult()
|
||||
r.set({"status": "fail", "error": "boom"})
|
||||
assert r.value == {"status": "fail", "error": "boom"}
|
||||
|
||||
|
||||
def test_set_not_applicable_requires_reason():
|
||||
r = CompatResult()
|
||||
with pytest.raises(ValueError, match="requires 'reason'"):
|
||||
r.set({"status": "not_applicable"})
|
||||
|
||||
|
||||
def test_set_not_applicable_with_reason_is_accepted():
|
||||
r = CompatResult()
|
||||
r.set({"status": "not_applicable", "reason": "Bedrock has no /thinking"})
|
||||
assert r.value == {"status": "not_applicable", "reason": "Bedrock has no /thinking"}
|
||||
|
||||
|
||||
def test_set_not_tested_is_accepted():
|
||||
r = CompatResult()
|
||||
r.set({"status": "not_tested"})
|
||||
assert r.value == {"status": "not_tested"}
|
||||
|
||||
|
||||
def test_set_rejects_unknown_status():
|
||||
r = CompatResult()
|
||||
with pytest.raises(ValueError, match="status must be one of"):
|
||||
r.set({"status": "maybe"})
|
||||
|
||||
|
||||
def test_set_rejects_non_dict():
|
||||
r = CompatResult()
|
||||
with pytest.raises(TypeError):
|
||||
r.set("pass") # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_set_copies_input():
|
||||
"""Mutating the dict after set() must not change the stored value."""
|
||||
r = CompatResult()
|
||||
payload = {"status": "fail", "error": "x"}
|
||||
r.set(payload)
|
||||
payload["error"] = "mutated"
|
||||
assert r.value["error"] == "x"
|
||||
|
|
@ -0,0 +1,98 @@
|
|||
"""basic_messaging_non_streaming × Anthropic.
|
||||
|
||||
The thinnest end-to-end path through every layer of the matrix: drive the
|
||||
real `claude` CLI in headless mode against a running LiteLLM proxy that
|
||||
routes to Anthropic, and report the outcome via `compat_result`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/basic_messaging_non_streaming/test_anthropic.py
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
|
||||
4.7; the cell only goes green if all three pass. We parametrize over the
|
||||
three models and the conftest aggregator produces one cell from the three
|
||||
results.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import ClaudeCLIError, run_claude
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Per the PRD: each cell is exercised against three Claude tiers via the
|
||||
# Anthropic provider. Aliases are configured in the LiteLLM proxy's
|
||||
# routing config; the driver only sends the alias.
|
||||
ANTHROPIC_MODELS = [
|
||||
"claude-haiku-4-5",
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ANTHROPIC_MODELS)
|
||||
def test_basic_messaging_non_streaming_anthropic(compat_result, model):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply.
|
||||
|
||||
"Basic messaging" means: send a single user prompt, receive any
|
||||
non-empty assistant text reply, no tools, no streaming, no thinking.
|
||||
The whole point of this slice is to prove the path works at all —
|
||||
so the assertion is intentionally lenient on the reply contents.
|
||||
"""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
try:
|
||||
result = run_claude(
|
||||
prompt="Reply with the single word 'pong' and nothing else.",
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
)
|
||||
except ClaudeCLIError as exc:
|
||||
compat_result.set({"status": "fail", "error": f"[{model}] {exc}"})
|
||||
pytest.fail(str(exc), pytrace=False)
|
||||
return
|
||||
|
||||
if result.exit_code != 0:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": f"[{model}] claude CLI exited {result.exit_code}: {result.stderr.strip()}",
|
||||
}
|
||||
)
|
||||
pytest.fail(f"claude CLI exited {result.exit_code} for {model}", pytrace=False)
|
||||
return
|
||||
|
||||
if not result.text.strip():
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": f"[{model}] claude returned empty assistant text",
|
||||
}
|
||||
)
|
||||
pytest.fail(f"empty reply for {model}", pytrace=False)
|
||||
return
|
||||
|
||||
compat_result.set({"status": "pass"})
|
||||
194
tests/claude_code/cli_driver.py
Normal file
194
tests/claude_code/cli_driver.py
Normal file
|
|
@ -0,0 +1,194 @@
|
|||
"""Claude Code CLI Driver.
|
||||
|
||||
A thin wrapper around the `claude` CLI in headless mode. Every compatibility
|
||||
test consumes only this module — tests must never shell out directly. This
|
||||
keeps the subprocess assembly, stream-JSON parsing, and result shape in a
|
||||
single place that can be unit-tested with a mocked subprocess.
|
||||
|
||||
The driver is deliberately small: it knows how to invoke the CLI, drain its
|
||||
stream-JSON output, and return a structured `DriverResult`. Higher-level
|
||||
matrix concerns (status aggregation, manifest lookup, JSON serialization)
|
||||
live in `matrix_builder.py`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Mapping, Optional, Sequence
|
||||
|
||||
CLAUDE_CLI_DEFAULT = "claude"
|
||||
DEFAULT_TIMEOUT_SECONDS = 120
|
||||
|
||||
|
||||
class ClaudeCLIError(RuntimeError):
|
||||
"""Raised when the `claude` CLI cannot be invoked or returns a fatal error."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class DriverResult:
|
||||
"""Structured outcome of a single `claude` CLI invocation.
|
||||
|
||||
`text` is the assistant's final user-visible reply (joined across any
|
||||
intermediate `assistant` events for non-streaming runs). `events` is the
|
||||
raw list of stream-JSON objects emitted by the CLI, preserved so test
|
||||
authors can write feature-specific assertions (tool calls, cache hits,
|
||||
usage) without re-parsing stdout.
|
||||
"""
|
||||
|
||||
text: str
|
||||
events: List[Dict[str, Any]] = field(default_factory=list)
|
||||
exit_code: int = 0
|
||||
stderr: str = ""
|
||||
usage: Optional[Dict[str, Any]] = None
|
||||
duration_ms: Optional[int] = None
|
||||
|
||||
|
||||
def run_claude(
|
||||
*,
|
||||
prompt: str,
|
||||
model: str,
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
extra_env: Optional[Mapping[str, str]] = None,
|
||||
extra_args: Optional[Sequence[str]] = None,
|
||||
cli_path: str = CLAUDE_CLI_DEFAULT,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
runner: Optional[Any] = None,
|
||||
) -> DriverResult:
|
||||
"""Invoke `claude` once in headless stream-JSON mode and return the result.
|
||||
|
||||
The CLI is pointed at a LiteLLM proxy via `ANTHROPIC_BASE_URL` /
|
||||
`ANTHROPIC_AUTH_TOKEN`, so the same code path exercises every provider
|
||||
column — only the model id and the proxy's routing differ between
|
||||
invocations.
|
||||
|
||||
`runner` is an injection seam used by the unit tests: by default we call
|
||||
`subprocess.run`, but the test suite swaps in a fake that yields canned
|
||||
stream-JSON. Production callers should never set it.
|
||||
"""
|
||||
if not prompt:
|
||||
raise ValueError("prompt must be a non-empty string")
|
||||
if not model:
|
||||
raise ValueError("model must be a non-empty string")
|
||||
if not base_url:
|
||||
raise ValueError("base_url must be a non-empty string")
|
||||
if not api_key:
|
||||
raise ValueError("api_key must be a non-empty string")
|
||||
|
||||
cmd: List[str] = [
|
||||
cli_path,
|
||||
"--print",
|
||||
"--output-format",
|
||||
"stream-json",
|
||||
"--verbose",
|
||||
"--model",
|
||||
model,
|
||||
prompt,
|
||||
]
|
||||
if extra_args:
|
||||
cmd.extend(extra_args)
|
||||
|
||||
env = {**os.environ, **(extra_env or {})}
|
||||
env["ANTHROPIC_BASE_URL"] = base_url
|
||||
env["ANTHROPIC_AUTH_TOKEN"] = api_key
|
||||
|
||||
run_fn = runner or subprocess.run
|
||||
try:
|
||||
completed = run_fn(
|
||||
cmd,
|
||||
env=env,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=timeout,
|
||||
check=False,
|
||||
)
|
||||
except FileNotFoundError as exc:
|
||||
raise ClaudeCLIError(
|
||||
f"claude CLI not found at {cli_path!r}; install with `npm i -g @anthropic-ai/claude-code`"
|
||||
) from exc
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise ClaudeCLIError(f"claude CLI timed out after {timeout}s") from exc
|
||||
|
||||
events = _parse_stream_json(completed.stdout or "")
|
||||
text = _extract_assistant_text(events)
|
||||
usage = _extract_usage(events)
|
||||
|
||||
return DriverResult(
|
||||
text=text,
|
||||
events=events,
|
||||
exit_code=completed.returncode,
|
||||
stderr=completed.stderr or "",
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
|
||||
def _parse_stream_json(stdout: str) -> List[Dict[str, Any]]:
|
||||
"""Parse newline-delimited JSON emitted by `claude --output-format stream-json`.
|
||||
|
||||
Lines that don't parse as JSON are silently skipped — the CLI occasionally
|
||||
emits debug output we don't care about, and a single malformed line should
|
||||
not abort the whole run. Real failure modes surface via exit code.
|
||||
"""
|
||||
events: List[Dict[str, Any]] = []
|
||||
for line in stdout.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
obj = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(obj, dict):
|
||||
events.append(obj)
|
||||
return events
|
||||
|
||||
|
||||
def _extract_assistant_text(events: Sequence[Mapping[str, Any]]) -> str:
|
||||
"""Concatenate the text content of every `assistant` event in order.
|
||||
|
||||
The non-streaming `--print` path emits a single `assistant` event whose
|
||||
`message.content` is a list of content blocks. We walk the blocks and
|
||||
join every `text` block — the CLI prints other block types (e.g.
|
||||
`tool_use`) which we ignore for the basic-messaging case.
|
||||
"""
|
||||
chunks: List[str] = []
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
message = event.get("message") or {}
|
||||
content = message.get("content")
|
||||
if isinstance(content, str):
|
||||
chunks.append(content)
|
||||
continue
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") == "text" and isinstance(block.get("text"), str):
|
||||
chunks.append(block["text"])
|
||||
return "".join(chunks)
|
||||
|
||||
|
||||
def _extract_usage(events: Sequence[Mapping[str, Any]]) -> Optional[Dict[str, Any]]:
|
||||
"""Return the most recent `usage` block seen on any event, if any.
|
||||
|
||||
The CLI surfaces token + cache usage on the final `result` event for
|
||||
non-streaming runs, but earlier events also carry partial usage in some
|
||||
versions; taking the last non-empty one is the safe default.
|
||||
"""
|
||||
last: Optional[Dict[str, Any]] = None
|
||||
for event in events:
|
||||
usage = event.get("usage")
|
||||
if isinstance(usage, dict) and usage:
|
||||
last = usage
|
||||
continue
|
||||
message = event.get("message")
|
||||
if isinstance(message, dict):
|
||||
inner = message.get("usage")
|
||||
if isinstance(inner, dict) and inner:
|
||||
last = inner
|
||||
return last
|
||||
164
tests/claude_code/conftest.py
Normal file
164
tests/claude_code/conftest.py
Normal file
|
|
@ -0,0 +1,164 @@
|
|||
"""Pytest plumbing for the Claude Code compatibility matrix.
|
||||
|
||||
Two responsibilities live here:
|
||||
|
||||
1. The `compat_result` fixture — the only API a test author needs to learn.
|
||||
Tests call `compat_result.set({"status": "pass"})` (or fail / not_applicable)
|
||||
to report their outcome as a tagged union. The fixture is per-test and
|
||||
stores the last value reported.
|
||||
|
||||
2. The `pytest_runtest_makereport` hook — captures each test's reported result,
|
||||
infers (feature, provider) from the file path, and writes a single
|
||||
`compat-results.json` artifact next to JUnit XML. The Matrix JSON Builder
|
||||
consumes this artifact to produce the published `compatibility-matrix.json`.
|
||||
|
||||
The (feature, provider) inference comes from the test file path: the parent
|
||||
directory name is the feature_id (matching `manifest.yaml`), and the file
|
||||
stem after the leading `test_` is the provider id. This avoids per-file
|
||||
metadata that drifts.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import pytest
|
||||
|
||||
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
|
||||
RESULTS_ARTIFACT_ENV = "COMPAT_RESULTS_PATH"
|
||||
DEFAULT_ARTIFACT_PATH = "compat-results.json"
|
||||
|
||||
|
||||
@dataclass
|
||||
class CompatResult:
|
||||
"""Per-test recorder for compatibility outcomes.
|
||||
|
||||
Tests interact only via `.set(...)`. `.value` is read by the
|
||||
`pytest_runtest_makereport` hook after the test body finishes.
|
||||
"""
|
||||
|
||||
value: Optional[Dict[str, Any]] = None
|
||||
|
||||
def set(self, result: Dict[str, Any]) -> None:
|
||||
if not isinstance(result, dict):
|
||||
raise TypeError("compat_result.set() requires a dict")
|
||||
status = result.get("status")
|
||||
if status not in VALID_STATUSES:
|
||||
raise ValueError(
|
||||
f"compat_result.set() status must be one of {sorted(VALID_STATUSES)}, "
|
||||
f"got {status!r}"
|
||||
)
|
||||
if status == "fail" and not result.get("error"):
|
||||
raise ValueError("compat_result.set({'status': 'fail'}) requires 'error'")
|
||||
if status == "not_applicable" and not result.get("reason"):
|
||||
raise ValueError(
|
||||
"compat_result.set({'status': 'not_applicable'}) requires 'reason'"
|
||||
)
|
||||
self.value = dict(result)
|
||||
|
||||
|
||||
@dataclass
|
||||
class _CollectedResult:
|
||||
feature_id: str
|
||||
provider: str
|
||||
nodeid: str
|
||||
result: Dict[str, Any]
|
||||
|
||||
|
||||
@dataclass
|
||||
class _Collector:
|
||||
items: List[_CollectedResult] = field(default_factory=list)
|
||||
|
||||
|
||||
_COLLECTOR = _Collector()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def compat_result() -> CompatResult:
|
||||
"""Per-test recorder for the (feature, provider) outcome.
|
||||
|
||||
Tests should call `compat_result.set({"status": "pass"})` (or fail /
|
||||
not_applicable) before returning. If a test exits without calling `.set()`
|
||||
the harness records `status="fail"` with an explanatory error so that
|
||||
every collected node maps to a real cell.
|
||||
"""
|
||||
return CompatResult()
|
||||
|
||||
|
||||
def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]:
|
||||
"""Infer (feature_id, provider) from a test file path.
|
||||
|
||||
Path shape: tests/claude_code/<feature_id>/test_<provider>.py
|
||||
Returns None if the file is not a per-feature test (e.g. unit tests
|
||||
living under tests/claude_code/_driver_unit_tests/), so those don't
|
||||
pollute the matrix artifact.
|
||||
"""
|
||||
name = node_path.name
|
||||
if not name.startswith("test_") or not name.endswith(".py"):
|
||||
return None
|
||||
provider = name[len("test_") : -len(".py")]
|
||||
feature_id = node_path.parent.name
|
||||
if feature_id.startswith("_") or feature_id == "claude_code":
|
||||
return None
|
||||
return feature_id, provider
|
||||
|
||||
|
||||
@pytest.hookimpl(hookwrapper=True)
|
||||
def pytest_runtest_makereport(item, call):
|
||||
"""Capture compat_result.value at end-of-test and remember it for the artifact."""
|
||||
outcome = yield
|
||||
report = outcome.get_result()
|
||||
if report.when != "call":
|
||||
return
|
||||
|
||||
inferred = _infer_feature_and_provider(Path(str(item.path)))
|
||||
if inferred is None:
|
||||
return
|
||||
feature_id, provider = inferred
|
||||
|
||||
fixture = item.funcargs.get("compat_result") if hasattr(item, "funcargs") else None
|
||||
reported: Optional[Dict[str, Any]] = getattr(fixture, "value", None)
|
||||
|
||||
if reported is None:
|
||||
if report.passed:
|
||||
reported = {
|
||||
"status": "fail",
|
||||
"error": "test passed without calling compat_result.set(); "
|
||||
"every compat test must report a status.",
|
||||
}
|
||||
else:
|
||||
reported = {
|
||||
"status": "fail",
|
||||
"error": (str(report.longrepr) if report.longrepr else "test failed"),
|
||||
}
|
||||
|
||||
_COLLECTOR.items.append(
|
||||
_CollectedResult(
|
||||
feature_id=feature_id,
|
||||
provider=provider,
|
||||
nodeid=report.nodeid,
|
||||
result=reported,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def pytest_sessionfinish(session, exitstatus):
|
||||
"""Write the structured results artifact at end of session."""
|
||||
artifact_path = os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH
|
||||
payload = {
|
||||
"schema_version": "1",
|
||||
"results": [
|
||||
{
|
||||
"feature_id": item.feature_id,
|
||||
"provider": item.provider,
|
||||
"nodeid": item.nodeid,
|
||||
"result": item.result,
|
||||
}
|
||||
for item in _COLLECTOR.items
|
||||
],
|
||||
}
|
||||
Path(artifact_path).write_text(json.dumps(payload, indent=2, sort_keys=True))
|
||||
26
tests/claude_code/manifest.yaml
Normal file
26
tests/claude_code/manifest.yaml
Normal file
|
|
@ -0,0 +1,26 @@
|
|||
# Claude Code Compatibility Matrix — feature manifest.
|
||||
#
|
||||
# Defines the row order of the matrix and maps each feature_id to its
|
||||
# human-readable display name. Adding a new feature to the matrix is a
|
||||
# three-step change:
|
||||
# 1. Append an entry to `features:` below.
|
||||
# 2. Create a directory `tests/claude_code/<feature_id>/`.
|
||||
# 3. Add per-provider test files inside that directory.
|
||||
#
|
||||
# `feature_id` MUST match the directory name on disk; the test harness
|
||||
# infers (feature, provider) for each test from its file path.
|
||||
|
||||
schema_version: "1"
|
||||
|
||||
# Provider column order in the rendered matrix.
|
||||
providers:
|
||||
- anthropic
|
||||
- bedrock_invoke
|
||||
- bedrock_converse
|
||||
- vertex_ai
|
||||
- azure
|
||||
|
||||
# Feature row order.
|
||||
features:
|
||||
- id: basic_messaging_non_streaming
|
||||
name: Basic messaging (non-streaming)
|
||||
179
tests/claude_code/matrix_builder.py
Normal file
179
tests/claude_code/matrix_builder.py
Normal file
|
|
@ -0,0 +1,179 @@
|
|||
"""Matrix JSON Builder.
|
||||
|
||||
Pure-function module that consumes the pytest-produced `compat-results.json`,
|
||||
the manifest, and run metadata, and emits the final `compatibility-matrix.json`
|
||||
conforming to the schema published in the PRD.
|
||||
|
||||
This module is deliberately free of subprocess, network, or filesystem side
|
||||
effects in its public API — the public entry points take pre-loaded inputs
|
||||
and return data structures, so they can be exercised by golden-file tests
|
||||
without I/O. A small `build_from_paths()` convenience wrapper does the I/O
|
||||
for callers that need it (the daily-cron publisher).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Mapping, Optional, Sequence
|
||||
|
||||
import yaml
|
||||
|
||||
SCHEMA_VERSION = "1"
|
||||
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
|
||||
|
||||
|
||||
class ManifestError(ValueError):
|
||||
"""Raised when `manifest.yaml` is malformed."""
|
||||
|
||||
|
||||
class ResultsError(ValueError):
|
||||
"""Raised when the pytest results artifact is malformed."""
|
||||
|
||||
|
||||
def load_manifest(path: Path) -> Dict[str, Any]:
|
||||
"""Load and validate `manifest.yaml`.
|
||||
|
||||
Returns a dict with keys: schema_version, providers, features. Raises
|
||||
ManifestError on missing fields or schema mismatch.
|
||||
"""
|
||||
raw = yaml.safe_load(path.read_text())
|
||||
if not isinstance(raw, dict):
|
||||
raise ManifestError(f"manifest at {path} is not a mapping")
|
||||
schema_version = str(raw.get("schema_version", ""))
|
||||
if schema_version != SCHEMA_VERSION:
|
||||
raise ManifestError(
|
||||
f"manifest schema_version {schema_version!r} does not match "
|
||||
f"builder version {SCHEMA_VERSION!r}"
|
||||
)
|
||||
providers = raw.get("providers")
|
||||
if not isinstance(providers, list) or not providers:
|
||||
raise ManifestError("manifest.providers must be a non-empty list")
|
||||
features = raw.get("features")
|
||||
if not isinstance(features, list) or not features:
|
||||
raise ManifestError("manifest.features must be a non-empty list")
|
||||
for feature in features:
|
||||
if not isinstance(feature, dict):
|
||||
raise ManifestError("each feature must be a mapping")
|
||||
if not feature.get("id") or not feature.get("name"):
|
||||
raise ManifestError("each feature must have id and name")
|
||||
return raw
|
||||
|
||||
|
||||
def load_results(path: Path) -> List[Dict[str, Any]]:
|
||||
"""Load the pytest results artifact and return its `results` list."""
|
||||
raw = json.loads(path.read_text())
|
||||
if not isinstance(raw, dict) or not isinstance(raw.get("results"), list):
|
||||
raise ResultsError(f"results artifact at {path} has no `results` list")
|
||||
return raw["results"]
|
||||
|
||||
|
||||
def build_matrix(
|
||||
*,
|
||||
manifest: Mapping[str, Any],
|
||||
results: Sequence[Mapping[str, Any]],
|
||||
litellm_version: str,
|
||||
claude_code_version: str,
|
||||
generated_at: str,
|
||||
) -> Dict[str, Any]:
|
||||
"""Build the published matrix JSON from pre-loaded inputs.
|
||||
|
||||
Empty cells (no test ran for a (feature, provider) and no
|
||||
`not_applicable` was declared) are filled in with `not_tested`. If
|
||||
multiple results report on the same cell — e.g. a per-feature test
|
||||
file containing one parametrize per Claude model — the cell aggregates
|
||||
to `pass` only if every model passed; otherwise `fail` with the first
|
||||
breaking model surfaced in the error.
|
||||
"""
|
||||
providers: List[str] = list(manifest["providers"])
|
||||
feature_specs: List[Dict[str, Any]] = list(manifest["features"])
|
||||
|
||||
grouped: Dict[tuple, List[Dict[str, Any]]] = {}
|
||||
for entry in results:
|
||||
if not isinstance(entry, Mapping):
|
||||
continue
|
||||
feature_id = entry.get("feature_id")
|
||||
provider = entry.get("provider")
|
||||
result = entry.get("result")
|
||||
if not feature_id or not provider or not isinstance(result, Mapping):
|
||||
continue
|
||||
if result.get("status") not in VALID_STATUSES:
|
||||
continue
|
||||
grouped.setdefault((feature_id, provider), []).append(dict(result))
|
||||
|
||||
features_out: List[Dict[str, Any]] = []
|
||||
for spec in feature_specs:
|
||||
feature_id = spec["id"]
|
||||
cells: Dict[str, Dict[str, Any]] = {}
|
||||
for provider in providers:
|
||||
cell_results = grouped.get((feature_id, provider), [])
|
||||
cells[provider] = _aggregate_cell(cell_results)
|
||||
features_out.append(
|
||||
{
|
||||
"id": feature_id,
|
||||
"name": spec["name"],
|
||||
"providers": cells,
|
||||
}
|
||||
)
|
||||
|
||||
return {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"generated_at": generated_at,
|
||||
"litellm_version": litellm_version,
|
||||
"claude_code_version": claude_code_version,
|
||||
"providers": providers,
|
||||
"features": features_out,
|
||||
}
|
||||
|
||||
|
||||
def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
|
||||
"""Aggregate a list of per-model results into a single cell status.
|
||||
|
||||
Order of precedence (most informative wins):
|
||||
- Any `fail` → cell is `fail` with the first failure's error.
|
||||
- `not_applicable` → cell is `not_applicable` with the reason.
|
||||
- `pass` → cell is `pass`.
|
||||
- empty / nothing recognized → `not_tested`.
|
||||
"""
|
||||
if not results:
|
||||
return {"status": "not_tested"}
|
||||
|
||||
for r in results:
|
||||
if r.get("status") == "fail":
|
||||
return {"status": "fail", "error": str(r.get("error", "test failed"))}
|
||||
|
||||
for r in results:
|
||||
if r.get("status") == "not_applicable":
|
||||
return {
|
||||
"status": "not_applicable",
|
||||
"reason": str(r.get("reason", "not applicable")),
|
||||
}
|
||||
|
||||
if all(r.get("status") == "pass" for r in results):
|
||||
return {"status": "pass"}
|
||||
|
||||
return {"status": "not_tested"}
|
||||
|
||||
|
||||
def build_from_paths(
|
||||
*,
|
||||
manifest_path: Path,
|
||||
results_path: Path,
|
||||
litellm_version: str,
|
||||
claude_code_version: str,
|
||||
generated_at: str,
|
||||
output_path: Optional[Path] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""I/O wrapper around build_matrix used by the publisher script."""
|
||||
manifest = load_manifest(manifest_path)
|
||||
results = load_results(results_path)
|
||||
matrix = build_matrix(
|
||||
manifest=manifest,
|
||||
results=results,
|
||||
litellm_version=litellm_version,
|
||||
claude_code_version=claude_code_version,
|
||||
generated_at=generated_at,
|
||||
)
|
||||
if output_path is not None:
|
||||
output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n")
|
||||
return matrix
|
||||
36
tests/claude_code/sample_compatibility-matrix.json
Normal file
36
tests/claude_code/sample_compatibility-matrix.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"schema_version": "1",
|
||||
"generated_at": "2026-04-25T00:00:00Z",
|
||||
"litellm_version": "v1.83.0-stable",
|
||||
"claude_code_version": "2.1.120",
|
||||
"providers": [
|
||||
"anthropic",
|
||||
"bedrock_invoke",
|
||||
"bedrock_converse",
|
||||
"vertex_ai",
|
||||
"azure"
|
||||
],
|
||||
"features": [
|
||||
{
|
||||
"id": "basic_messaging_non_streaming",
|
||||
"name": "Basic messaging (non-streaming)",
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"status": "pass"
|
||||
},
|
||||
"bedrock_invoke": {
|
||||
"status": "not_tested"
|
||||
},
|
||||
"bedrock_converse": {
|
||||
"status": "not_tested"
|
||||
},
|
||||
"vertex_ai": {
|
||||
"status": "not_tested"
|
||||
},
|
||||
"azure": {
|
||||
"status": "not_tested"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue