litellm/tests/claude_code/matrix_builder.py
Cursor Agent 9928da27f3
fix(claude_code): bugbot — aggregate all fail errors + structural test every manifest feature
Addresses three Bugbot concerns flagged on PR #28027 that are real
behavioral / coverage gaps:

1. matrix_builder._aggregate_cell now joins every failing tier's error
   in the published cell instead of silently dropping all but the first.
   When Haiku 429s and Opus times out on the same cell, both diagnostics
   land in the matrix JSON so docs-page triage can name both outliers.

2. _aggregate_cell treats 'not_tested' rows as absent data: they're
   dropped before computing the cell status. Previously a mixed
   (pass, not_tested) cell silently fell through to 'not_tested',
   discarding the passing tiers and hiding real coverage from the
   published matrix. A cell still aggregates to 'not_tested' when
   *every* row is 'not_tested' (or there are no rows at all).

3. test_v0_layout.py now structurally validates every feature declared
   in manifest.yaml (directory exists, __init__.py exists, every
   per-provider test_<provider>.py exists), not just the original six
   v0 rows. The EXPECTED_FEATURE_IDS / EXPECTED_PROVIDERS anchor
   constants still pin v0 positions; the new manifest-driven tests
   extend the same structural guarantees to every post-v0 row so a
   broken directory in 'count_tokens', 'tool_search', 'web_search',
   etc. fails CI instead of silently becoming a 'not_tested' cell.

Three new builder tests pin the new aggregation behavior:
  - mixed pass + not_tested surfaces as pass
  - all-not_tested stays not_tested
  - multiple fail errors are joined with '; '

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
2026-05-18 03:48:23 +00:00

193 lines
7 KiB
Python

"""Matrix JSON Builder.
Pure-function module that consumes the pytest-produced `compat-results.json`,
the manifest, and run metadata, and emits the final `compatibility-matrix.json`
conforming to the schema published in the PRD.
This module is deliberately free of subprocess, network, or filesystem side
effects in its public API — the public entry points take pre-loaded inputs
and return data structures, so they can be exercised by golden-file tests
without I/O. A small `build_from_paths()` convenience wrapper does the I/O
for callers that need it (the daily-cron publisher).
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Dict, List, Mapping, Optional, Sequence
import yaml
SCHEMA_VERSION = "1"
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
class ManifestError(ValueError):
"""Raised when `manifest.yaml` is malformed."""
class ResultsError(ValueError):
"""Raised when the pytest results artifact is malformed."""
def load_manifest(path: Path) -> Dict[str, Any]:
"""Load and validate `manifest.yaml`.
Returns a dict with keys: schema_version, providers, features. Raises
ManifestError on missing fields or schema mismatch.
"""
raw = yaml.safe_load(path.read_text())
if not isinstance(raw, dict):
raise ManifestError(f"manifest at {path} is not a mapping")
schema_version = str(raw.get("schema_version", ""))
if schema_version != SCHEMA_VERSION:
raise ManifestError(
f"manifest schema_version {schema_version!r} does not match "
f"builder version {SCHEMA_VERSION!r}"
)
providers = raw.get("providers")
if not isinstance(providers, list) or not providers:
raise ManifestError("manifest.providers must be a non-empty list")
features = raw.get("features")
if not isinstance(features, list) or not features:
raise ManifestError("manifest.features must be a non-empty list")
for feature in features:
if not isinstance(feature, dict):
raise ManifestError("each feature must be a mapping")
if not feature.get("id") or not feature.get("name"):
raise ManifestError("each feature must have id and name")
return raw
def load_results(path: Path) -> List[Dict[str, Any]]:
"""Load the pytest results artifact and return its `results` list."""
raw = json.loads(path.read_text())
if not isinstance(raw, dict) or not isinstance(raw.get("results"), list):
raise ResultsError(f"results artifact at {path} has no `results` list")
return raw["results"]
def build_matrix(
*,
manifest: Mapping[str, Any],
results: Sequence[Mapping[str, Any]],
litellm_version: str,
claude_code_version: str,
generated_at: str,
) -> Dict[str, Any]:
"""Build the published matrix JSON from pre-loaded inputs.
Empty cells (no test ran for a (feature, provider) and no
`not_applicable` was declared) are filled in with `not_tested`. If
multiple results report on the same cell — e.g. a per-feature test
file containing one parametrize per Claude model — the cell aggregates
to `pass` only if every model passed; otherwise `fail` with the first
breaking model surfaced in the error.
"""
providers: List[str] = list(manifest["providers"])
feature_specs: List[Dict[str, Any]] = list(manifest["features"])
grouped: Dict[tuple, List[Dict[str, Any]]] = {}
for entry in results:
if not isinstance(entry, Mapping):
continue
feature_id = entry.get("feature_id")
provider = entry.get("provider")
result = entry.get("result")
if not feature_id or not provider or not isinstance(result, Mapping):
continue
if result.get("status") not in VALID_STATUSES:
continue
grouped.setdefault((feature_id, provider), []).append(dict(result))
features_out: List[Dict[str, Any]] = []
for spec in feature_specs:
feature_id = spec["id"]
cells: Dict[str, Dict[str, Any]] = {}
for provider in providers:
cell_results = grouped.get((feature_id, provider), [])
cells[provider] = _aggregate_cell(cell_results)
features_out.append(
{
"id": feature_id,
"name": spec["name"],
"providers": cells,
}
)
return {
"schema_version": SCHEMA_VERSION,
"generated_at": generated_at,
"litellm_version": litellm_version,
"claude_code_version": claude_code_version,
"providers": providers,
"features": features_out,
}
def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
"""Aggregate a list of per-model results into a single cell status.
Order of precedence (most informative wins):
- Any `fail` → cell is `fail` with every failing model's error
joined by `"; "` so a multi-tier breakage doesn't silently hide
all but the first error from the published matrix.
- `not_applicable` → cell is `not_applicable` with the reason.
- `pass` → cell is `pass`.
- empty / nothing recognized → `not_tested`.
`not_tested` rows are treated as absent data: they're dropped before
aggregation so a mix of (pass, not_tested) — e.g. from a partial
crash or a test that explicitly recorded "this tier didn't run"
still surfaces the passing tiers rather than silently demoting the
whole cell to `not_tested`. A cell is only `not_tested` when *every*
row is `not_tested` (or there are no rows at all).
"""
if not results:
return {"status": "not_tested"}
observed = [r for r in results if r.get("status") != "not_tested"]
if not observed:
return {"status": "not_tested"}
failures = [r for r in observed if r.get("status") == "fail"]
if failures:
errors = [str(r.get("error", "test failed")) for r in failures]
return {"status": "fail", "error": "; ".join(errors)}
for r in observed:
if r.get("status") == "not_applicable":
return {
"status": "not_applicable",
"reason": str(r.get("reason", "not applicable")),
}
if all(r.get("status") == "pass" for r in observed):
return {"status": "pass"}
return {"status": "not_tested"}
def build_from_paths(
*,
manifest_path: Path,
results_path: Path,
litellm_version: str,
claude_code_version: str,
generated_at: str,
output_path: Optional[Path] = None,
) -> Dict[str, Any]:
"""I/O wrapper around build_matrix used by the publisher script."""
manifest = load_manifest(manifest_path)
results = load_results(results_path)
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version=litellm_version,
claude_code_version=claude_code_version,
generated_at=generated_at,
)
if output_path is not None:
output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n")
return matrix