test(claude_code): move the Claude Code compatibility matrix under tests/e2e (#32548)

* test(claude_code): move the Claude Code compatibility matrix under tests/e2e

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* ci(claude_code): drop the CircleCI compat PR gate; the matrix runs in the scheduled e2e suite instead

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* ci: restore the upload-coverage job dropped by mistake with the compat gate

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(e2e/claude_code): print rate-limit summary on failed compat runs and fix stale run_daily.sh header comments

* test(claude_code): assert fine-grained tool streaming via input_json_delta instead of an event-count floor

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: mateo <mateo@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-07-14 19:19:03 -07:00 committed by GitHub
parent 65ca095d4d
commit 56f4dbf60a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
130 changed files with 16288 additions and 1 deletions

7
.gitignore vendored
View file

@ -106,6 +106,13 @@ STABILIZATION_TODO.md
**/coverage
test-config
# Claude Code compatibility-matrix pytest artifact (CI-only output).
compat-results.json
compat-results.json.shards/
compat-rate-limit-summary.json
# Matrix JSON produced by the daily-cron publisher (pushed to litellm-docs).
compatibility-matrix.json
# ---------- Terraform ----------
# Provider binaries + module cache — regenerated by `terraform init`.
**/.terraform/

View file

@ -1,7 +1,7 @@
{
"include": ["litellm"],
"ignore": [],
"exclude": ["**/node_modules", "**/__pycache__", "litellm/types/utils.py", "litellm/proxy/_types.py"],
"exclude": ["**/node_modules", "**/__pycache__", "tests/e2e/claude_code", "litellm/types/utils.py", "litellm/proxy/_types.py"],
"pythonVersion": "3.12",
"typeCheckingMode": "strict",
"enableTypeIgnoreComments": false,

View file

@ -17,6 +17,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
- `security/` - secret handling and log-leak protection
- `router/` - routing and reliability behavior (fallbacks, cooldowns)
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees, and does not use the shared transport harness
## Lay the pattern down in a class

View file

View file

@ -0,0 +1,167 @@
"""Shared body for the `basic_messaging_*` × <provider> compat cells.
Every basic_messaging cell follows the same skeleton:
1. Read the proxy base URL + API key from env, fail-early if missing.
2. Fan the three Claude tiers out via `run_claude_models_parallel`.
3. Inspect each model's outcome and report one `compat_result` row per
model `ClaudeCLIError`, non-zero exit, and empty assistant text
are all per-model fails; everything else is a per-model pass.
4. Surface a joined failure message via `pytest.fail(...)` so the
pytest run also goes red.
The streaming variant additionally passes `verify_streaming=True`,
which adds the `--include-partial-messages` CLI flag and asserts that
the proxy actually streamed the response (see the helper docstring for
the wire-level rationale).
The conftest infers `(feature_id, provider)` purely from the test file
path, so each per-provider file just declares its model list and calls
`run_basic_messaging_cell(...)`. This keeps all cell logic in one place
a future tweak to the env-missing guard or the failure-loop shape
now propagates to every cell automatically.
The leading underscore in the filename is what keeps pytest from
collecting this module as a test file.
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Floor on the number of `stream_event` records (with delta payloads)
# we expect to see when the proxy actually streams. With
# `--include-partial-messages`, the CLI emits one `stream_event` per
# raw upstream SSE event — a fully-streamed response produces many
# (`message_start`, multiple `content_block_delta`s, `content_block_stop`,
# `message_delta`, `message_stop`); a proxy that buffers the upstream
# and returns a single non-streaming chunk produces 0 or 1. Floor of 2
# is safely above the buffered case for any non-trivial reply, which
# is why the streaming cells use a "count from 1 to 5" style prompt.
MIN_STREAM_DELTA_EVENTS = 2
def _count_stream_event_deltas(events: Sequence[Mapping[str, Any]]) -> int:
"""Count `stream_event` records that carry an SSE event payload.
With `--include-partial-messages`, Claude Code wraps every upstream
SSE event in a `{"type": "stream_event", "event": {...}}` record.
A buffering proxy collapses the upstream stream into a single
non-streaming response, so these records vanish. Counting them
(rather than just `len(events)`) is the wire-level signal that
"did the proxy preserve streaming?" independent of the `system`
/`assistant`/`result` boilerplate records the CLI always emits.
"""
count = 0
for event in events:
if event.get("type") != "stream_event":
continue
if isinstance(event.get("event"), Mapping):
count += 1
return count
def run_basic_messaging_cell(
*,
compat_result,
models: Sequence[str],
prompt: str,
verify_streaming: bool = False,
) -> None:
"""Run the shared `basic_messaging_*` × <provider> cell body.
When ``verify_streaming=True``, the cell additionally asserts that
the proxy streamed the response end-to-end. The check works by
passing ``--include-partial-messages`` to the `claude` CLI, which
causes it to emit one ``stream_event`` record per raw upstream SSE
event (``message_start``, ``content_block_delta``, ``message_stop``,
etc.). A proxy that buffers the upstream stream and returns a
single non-streaming response collapses those records to zero
so a floor of ``MIN_STREAM_DELTA_EVENTS`` ``stream_event`` records
catches the buffering regression without needing a streaming-aware
driver.
This is the same shape of check as ``tool_use_streaming`` uses,
just keyed off the explicit partial-message flag so it works for
plain assistant replies (where the CLI would otherwise collapse a
streamed reply to a single ``assistant`` event in
``--print --output-format stream-json`` mode).
"""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
extra_args: Sequence[str] = (
("--include-partial-messages",) if verify_streaming else ()
)
outcomes = run_claude_models_parallel(
models=models,
prompt=prompt,
base_url=base_url,
api_key=api_key,
extra_args=extra_args,
)
failures = []
for model in models:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if verify_streaming:
stream_event_count = _count_stream_event_deltas(outcome.events)
if stream_event_count < MIN_STREAM_DELTA_EVENTS:
error = (
f"[{model}] only {stream_event_count} stream_event records "
f"observed (< {MIN_STREAM_DELTA_EVENTS}); proxy likely "
f"buffered the upstream response instead of streaming it"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,38 @@
{
"schema_version": "1",
"generated_at": "2026-04-25T00:00:00Z",
"litellm_version": "v1.83.0-stable",
"claude_code_version": "2.1.120",
"providers": [
"anthropic",
"bedrock_invoke"
],
"features": [
{
"id": "basic_messaging_non_streaming",
"name": "Basic messaging (non-streaming)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "not_tested"
}
}
},
{
"id": "tool_use",
"name": "Tool use",
"providers": {
"anthropic": {
"status": "fail",
"error": "[claude-sonnet-4-6] tool call dropped"
},
"bedrock_invoke": {
"status": "not_applicable",
"reason": "tool use not yet wired up for Bedrock Invoke"
}
}
}
]
}

View file

@ -0,0 +1,9 @@
schema_version: "1"
providers:
- anthropic
- bedrock_invoke
features:
- id: basic_messaging_non_streaming
name: Basic messaging (non-streaming)
- id: tool_use
name: Tool use

View file

@ -0,0 +1,41 @@
{
"schema_version": "1",
"results": [
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-haiku-4-5]",
"result": {"status": "pass"}
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-sonnet-4-6]",
"result": {"status": "pass"}
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "anthropic",
"nodeid": "tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py::test_basic_messaging_non_streaming_anthropic[claude-opus-4-7]",
"result": {"status": "pass"}
},
{
"feature_id": "tool_use",
"provider": "anthropic",
"nodeid": "tests/e2e/claude_code/tool_use/test_anthropic.py::test_x[claude-haiku-4-5]",
"result": {"status": "pass"}
},
{
"feature_id": "tool_use",
"provider": "anthropic",
"nodeid": "tests/e2e/claude_code/tool_use/test_anthropic.py::test_x[claude-sonnet-4-6]",
"result": {"status": "fail", "error": "[claude-sonnet-4-6] tool call dropped"}
},
{
"feature_id": "tool_use",
"provider": "bedrock_invoke",
"nodeid": "tests/e2e/claude_code/tool_use/test_bedrock_invoke.py::test_x[claude-haiku-4-5]",
"result": {"status": "not_applicable", "reason": "tool use not yet wired up for Bedrock Invoke"}
}
]
}

View file

@ -0,0 +1,479 @@
"""Golden-file tests for the Matrix JSON Builder.
These tests fix the published JSON schema. The builder is a pure function
from (manifest, results, metadata) matrix dict, so we feed it a fixture
input set and compare the produced dict to a checked-in expected output.
Any schema drift intentional or accidental surfaces as a diff in PR
review.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from claude_code.matrix_builder import (
ManifestError,
ResultsError,
build_from_paths,
build_matrix,
load_manifest,
load_results,
)
FIXTURES = Path(__file__).parent / "fixtures"
def test_build_matrix_matches_golden_file(tmp_path):
manifest = load_manifest(FIXTURES / "manifest.yaml")
results = load_results(FIXTURES / "results.json")
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v1.83.0-stable",
claude_code_version="2.1.120",
generated_at="2026-04-25T00:00:00Z",
)
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
assert matrix == expected
def test_build_matrix_pass_requires_all_models_pass():
"""Multiple results in one cell must all be pass for the cell to be pass."""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
def test_build_matrix_any_fail_makes_cell_fail():
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "fail", "error": "[claude-opus-4-7] timeout"},
},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cell = matrix["features"][0]["providers"]["anthropic"]
assert cell["status"] == "fail"
assert cell["error"] == "[claude-opus-4-7] timeout"
def test_build_matrix_joins_all_failure_errors_in_one_cell():
"""When multiple tiers fail for different reasons within the same cell,
every failure's error must appear in the published cell so triage
isn't reduced to a single tier's diagnostic.
"""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "fail", "error": "[claude-haiku-4-5] 429"},
},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "fail", "error": "[claude-opus-4-7] timeout"},
},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cell = matrix["features"][0]["providers"]["anthropic"]
assert cell["status"] == "fail"
assert "[claude-haiku-4-5] 429" in cell["error"]
assert "[claude-opus-4-7] timeout" in cell["error"]
def test_build_matrix_mixed_pass_and_not_tested_surfaces_pass():
"""A `not_tested` row mixed with `pass` rows must not silently demote
the cell to `not_tested` `not_tested` is "absent data", not a
negative signal. Otherwise a partial crash mid-test, or a test that
explicitly recorded "tier didn't run", would discard real passing
results from the published cell.
"""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "not_tested"},
},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
def test_build_matrix_all_not_tested_stays_not_tested():
"""A cell whose every row is `not_tested` (or empty) must remain
`not_tested` the absent-data rule only drops `not_tested` rows
when there's other signal to surface.
"""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "not_tested"},
},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "not_tested"},
},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "not_tested"}
def test_build_matrix_mixed_pass_and_not_applicable_surfaces_pass():
"""A `not_applicable` row mixed with `pass` rows must surface as
`pass`, not `not_applicable`. The published cell answers "does this
feature work on this provider?"; if any tier passes, the feature
works there. Discarding passing tiers because one tier is NA would
misrepresent the cell as unsupported.
"""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
{
"feature_id": "f",
"provider": "anthropic",
"result": {
"status": "not_applicable",
"reason": "haiku does not support extended thinking",
},
},
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {"status": "pass"}
def test_build_matrix_all_not_applicable_stays_not_applicable():
"""When every observed row is `not_applicable`, the cell remains
`not_applicable` and the first row's reason carries through to the
published matrix.
"""
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{
"feature_id": "f",
"provider": "anthropic",
"result": {
"status": "not_applicable",
"reason": "feature unsupported on this provider",
},
},
{
"feature_id": "f",
"provider": "anthropic",
"result": {"status": "not_applicable", "reason": "ditto"},
},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["features"][0]["providers"]["anthropic"] == {
"status": "not_applicable",
"reason": "feature unsupported on this provider",
}
def test_build_matrix_fills_not_tested_for_missing_cells():
manifest = {
"schema_version": "1",
"providers": ["anthropic", "azure"],
"features": [{"id": "f", "name": "F"}],
}
results = [
{"feature_id": "f", "provider": "anthropic", "result": {"status": "pass"}},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cells = matrix["features"][0]["providers"]
assert cells["anthropic"] == {"status": "pass"}
assert cells["azure"] == {"status": "not_tested"}
def test_build_matrix_preserves_provider_and_feature_order():
manifest = {
"schema_version": "1",
"providers": ["azure", "anthropic", "vertex_ai"],
"features": [
{"id": "z", "name": "Z"},
{"id": "a", "name": "A"},
],
}
matrix = build_matrix(
manifest=manifest,
results=[],
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["providers"] == ["azure", "anthropic", "vertex_ai"]
assert [f["id"] for f in matrix["features"]] == ["z", "a"]
assert list(matrix["features"][0]["providers"].keys()) == [
"azure",
"anthropic",
"vertex_ai",
]
def test_build_matrix_emits_schema_version_one():
manifest = {
"schema_version": "1",
"providers": ["anthropic"],
"features": [{"id": "f", "name": "F"}],
}
matrix = build_matrix(
manifest=manifest,
results=[],
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
assert matrix["schema_version"] == "1"
def test_load_manifest_rejects_wrong_schema_version(tmp_path):
bad = tmp_path / "manifest.yaml"
bad.write_text(
'schema_version: "2"\nproviders: [anthropic]\nfeatures:\n - id: f\n name: F\n'
)
with pytest.raises(ManifestError, match="schema_version"):
load_manifest(bad)
def test_load_manifest_rejects_empty_features(tmp_path):
bad = tmp_path / "manifest.yaml"
bad.write_text('schema_version: "1"\nproviders: [anthropic]\nfeatures: []\n')
with pytest.raises(ManifestError):
load_manifest(bad)
def test_load_results_rejects_missing_results_key(tmp_path):
bad = tmp_path / "results.json"
bad.write_text(json.dumps({"schema_version": "1"}))
with pytest.raises(ResultsError):
load_results(bad)
def test_build_matrix_6x5_grid_matches_published_sample():
"""Slice 5 acceptance: feeding the per-model results the full v0
row set produces reproduces the hand-authored 6x5 sample that the
docs page renders.
Inputs mirror the structure of `compat-results.json` after a real
run with the proxy configured for all five columns and all six
feature directories: every (feature, provider, model) cell yields a
`pass`. Anthropic announced Claude in Microsoft Foundry on
2025-11-18, so the Azure column is now exercised end-to-end like
the others rather than reporting `not_applicable`.
The aggregated matrix must equal the checked-in
`sample_compatibility-matrix.json` byte-for-byte (after JSON load),
so any future schema drift surfaces here in review.
"""
repo_root = Path(__file__).resolve().parents[1]
full_manifest = load_manifest(repo_root / "manifest.yaml")
# The v0 sample matrix is a frozen baseline: it covers exactly the
# six features the PRD shipped with, in their canonical order. The
# live manifest may carry additional rows (extensions added after
# v0 shipped), but the sample is derived only from the v0 slice so
# this test stays a meaningful regression gate for the v0 cell
# shape rather than chasing every new row added downstream.
v0_feature_ids = [
"basic_messaging_non_streaming",
"basic_messaging_streaming",
"tool_use",
"prompt_caching_5m",
"vision",
# Row 6 of the v0 PRD; originally shipped as `extended_thinking`.
# The id was renamed in-place to `thinking` to match Anthropic's
# current docs (which reserve "extended thinking" for the
# deprecated manual mode only). The row's *position* in v0 is
# the load-bearing invariant, not the id string.
"thinking",
]
v0_features = [
feature
for feature in full_manifest["features"]
if feature["id"] in v0_feature_ids
]
manifest = {**full_manifest, "features": v0_features}
feature_ids = [feature["id"] for feature in manifest["features"]]
providers = manifest["providers"]
models = ["claude-haiku-4-5", "claude-sonnet-4-6", "claude-opus-4-7"]
results = []
for feature_id in feature_ids:
for provider in providers:
for model in models:
results.append(
{
"feature_id": feature_id,
"provider": provider,
"nodeid": (
f"tests/e2e/claude_code/{feature_id}/test_{provider}.py"
f"::test[{model}]"
),
"result": {"status": "pass"},
}
)
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v1.83.0-stable",
claude_code_version="2.1.120",
generated_at="2026-04-25T00:00:00Z",
)
expected = json.loads((repo_root / "sample_compatibility-matrix.json").read_text())
assert matrix == expected
def test_build_matrix_1x5_grid_one_failing_model_breaks_cell():
"""If even one of three models fails on a provider, that cell is fail
and the error string carries the failing model id so the docs
tooltip can name the outlier."""
repo_root = Path(__file__).resolve().parents[1]
manifest = load_manifest(repo_root / "manifest.yaml")
results = [
{
"feature_id": "basic_messaging_non_streaming",
"provider": "bedrock_invoke",
"result": {"status": "pass"},
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "bedrock_invoke",
"result": {
"status": "fail",
"error": "[claude-opus-4-7-bedrock-invoke] claude CLI exited 1: throttled",
},
},
{
"feature_id": "basic_messaging_non_streaming",
"provider": "bedrock_invoke",
"result": {"status": "pass"},
},
]
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version="v",
claude_code_version="c",
generated_at="t",
)
cell = matrix["features"][0]["providers"]["bedrock_invoke"]
assert cell["status"] == "fail"
assert "claude-opus-4-7-bedrock-invoke" in cell["error"]
def test_build_from_paths_writes_output(tmp_path):
out = tmp_path / "compatibility-matrix.json"
matrix = build_from_paths(
manifest_path=FIXTURES / "manifest.yaml",
results_path=FIXTURES / "results.json",
litellm_version="v1.83.0-stable",
claude_code_version="2.1.120",
generated_at="2026-04-25T00:00:00Z",
output_path=out,
)
assert out.exists()
on_disk = json.loads(out.read_text())
assert on_disk == matrix
expected = json.loads((FIXTURES / "expected_matrix.json").read_text())
assert on_disk == expected

View file

@ -0,0 +1,183 @@
"""Structural tests for the full v0 6x5 matrix layout.
These tests don't run the `claude` CLI — they only verify that the
shape of the test suite on disk matches what the PRD declares: six
features in the prescribed order, and for each feature a directory
with one test file per provider column.
Catching layout drift here means the daily-cron VM and the PR gate
both see the same row set the docs page declares.
"""
from __future__ import annotations
from pathlib import Path
import pytest
import yaml
REPO_ROOT = Path(__file__).resolve().parents[1]
MANIFEST_PATH = REPO_ROOT / "manifest.yaml"
# The PRD's "Features in v0" section, in row order.
EXPECTED_FEATURE_IDS = [
"basic_messaging_non_streaming",
"basic_messaging_streaming",
"tool_use",
"prompt_caching_5m",
"vision",
# v0 originally shipped this row as `extended_thinking`. It was
# renamed in-place to `thinking` because Anthropic's docs reserve
# "extended thinking" for the deprecated manual API mode only; the
# single row exercises both manual and adaptive shapes since Claude
# Code picks per model. The PRD's "v0" identity is the *position*
# (row 6, 0-indexed 5), not the id string.
"thinking",
]
# The PRD's column order. Every feature directory must have one
# `test_<provider>.py` for each of these.
EXPECTED_PROVIDERS = [
"anthropic",
"bedrock_invoke",
"bedrock_converse",
"vertex_ai",
"azure",
]
def _all_manifest_feature_ids() -> list[str]:
"""Every feature_id currently declared in `manifest.yaml`.
Evaluated at import time so the result can drive parametrized
structural tests below. Used to catch layout drift on post-v0
feature rows added after the matrix shipped the v0 anchor
constants above only validate the original six rows by design.
"""
return [
feature["id"]
for feature in yaml.safe_load(MANIFEST_PATH.read_text())["features"]
]
ALL_FEATURE_IDS = _all_manifest_feature_ids()
@pytest.fixture(scope="module")
def manifest() -> dict:
return yaml.safe_load(MANIFEST_PATH.read_text())
def test_manifest_lists_all_six_v0_features_in_order(manifest):
"""The PRD's v0 row set must appear at the top of the manifest in
order. Features beyond v0 (extensions added after the matrix
shipped) are allowed but must not reorder or displace the v0
rows the docs page anchors row links by index, so v0 stays
pinned at positions [0:6] for the lifetime of the schema.
"""
ids = [feature["id"] for feature in manifest["features"]]
assert ids[: len(EXPECTED_FEATURE_IDS)] == EXPECTED_FEATURE_IDS
def test_manifest_lists_all_five_v0_providers_in_order(manifest):
assert manifest["providers"] == EXPECTED_PROVIDERS
def test_manifest_every_feature_has_human_readable_name(manifest):
for feature in manifest["features"]:
assert isinstance(feature["name"], str) and feature["name"].strip()
@pytest.mark.parametrize("feature_id", EXPECTED_FEATURE_IDS)
def test_feature_directory_exists(feature_id):
feature_dir = REPO_ROOT / feature_id
assert feature_dir.is_dir(), f"missing feature directory: {feature_dir}"
@pytest.mark.parametrize("feature_id", EXPECTED_FEATURE_IDS)
@pytest.mark.parametrize("provider", EXPECTED_PROVIDERS)
def test_per_provider_test_file_exists(feature_id, provider):
test_file = REPO_ROOT / feature_id / f"test_{provider}.py"
assert test_file.is_file(), f"missing per-provider test file: {test_file}"
@pytest.mark.parametrize("feature_id", EXPECTED_FEATURE_IDS)
def test_feature_directory_has_init_file(feature_id):
"""Each feature directory needs an __init__.py so pytest collects
the per-provider test files as a package matches the layout
established by `basic_messaging_non_streaming/`."""
init_file = REPO_ROOT / feature_id / "__init__.py"
assert init_file.is_file(), f"missing __init__.py: {init_file}"
# Manifest-driven structural tests: every feature in `manifest.yaml`
# (v0 and post-v0 alike) must have the expected on-disk layout. The
# v0-only tests above pin the position of the original six rows; these
# extend the same structural guarantees to any row added afterward so
# a broken post-v0 directory still fails CI.
@pytest.mark.parametrize("feature_id", ALL_FEATURE_IDS)
def test_every_manifest_feature_has_directory(feature_id):
feature_dir = REPO_ROOT / feature_id
assert feature_dir.is_dir(), (
f"manifest declares {feature_id!r} but {feature_dir} is missing — "
"feature_id MUST match its on-disk directory (see manifest.yaml header)."
)
@pytest.mark.parametrize("feature_id", ALL_FEATURE_IDS)
def test_every_manifest_feature_has_init_file(feature_id):
init_file = REPO_ROOT / feature_id / "__init__.py"
assert init_file.is_file(), f"missing __init__.py: {init_file}"
@pytest.mark.parametrize("feature_id", ALL_FEATURE_IDS)
@pytest.mark.parametrize("provider", EXPECTED_PROVIDERS)
def test_every_manifest_feature_has_per_provider_test_file(feature_id, provider):
"""Every (feature, provider) cell in the rendered matrix must be
backed by a per-provider test file. Without this check, a missing
file silently becomes a `not_tested` cell in the published matrix
rather than a CI failure surfacing the layout drift."""
test_file = REPO_ROOT / feature_id / f"test_{provider}.py"
assert test_file.is_file(), f"missing per-provider test file: {test_file}"
@pytest.mark.parametrize("feature_id", EXPECTED_FEATURE_IDS)
@pytest.mark.parametrize("provider", EXPECTED_PROVIDERS)
def test_per_provider_test_file_imports_and_parametrizes_three_models(
feature_id, provider
):
"""Every test file must reference the three Claude tiers required
by the PRD: Haiku 4.5, Sonnet 4.6, Opus 4.7. Implementations may
use plain aliases or per-provider-suffixed aliases (e.g.
`claude-opus-4-7-bedrock-invoke`), so we check for the tier
substrings rather than exact alias names."""
text = (REPO_ROOT / feature_id / f"test_{provider}.py").read_text()
for tier in ("haiku-4-5", "sonnet-4-6", "opus-4-7"):
assert (
tier in text
), f"{feature_id}/test_{provider}.py does not reference {tier}"
@pytest.mark.parametrize("feature_id", EXPECTED_FEATURE_IDS)
def test_azure_test_file_drives_the_proxy(feature_id):
"""Azure (Microsoft Foundry) hosts Anthropic Claude as of 2025-11-18,
so every Azure cell in the v0 matrix exercises a real route through
the LiteLLM proxy same shape as the other provider columns. Pin
that here so a future regression doesn't silently revert these
cells to the old `not_applicable` boilerplate.
We accept either the direct `run_claude(...)` family of entrypoints
or a per-feature shared helper (e.g. `run_basic_messaging_cell`)
that wraps them both shapes drive the proxy, and we don't want
this layout pin to block legitimate de-duplication of test bodies.
"""
text = (REPO_ROOT / feature_id / "test_azure.py").read_text()
assert "run_claude" in text or "run_basic_messaging_cell" in text, (
f"{feature_id}/test_azure.py must drive the claude CLI via run_claude() "
"or a shared helper that wraps it; the not_applicable stub was removed "
"when Foundry started hosting Claude."
)
assert '"status": "not_applicable"' not in text, (
f"{feature_id}/test_azure.py still reports not_applicable; Microsoft Foundry "
"now hosts Claude (Haiku 4.5, Sonnet 4.6, Opus 4.7), so this row must run."
)

View file

@ -0,0 +1,32 @@
"""Local conftest for the driver unit tests.
Installs a hermetic, no-op rate limiter for every test in this
subdirectory. Without this, importing `cli_driver` and calling
`run_claude(..., runner=fake)` would silently consume tokens from the
shared default limiter (which writes to `$TMPDIR/...`), polluting the
on-disk state another test run might rely on and adding flakiness if
the env vars say "rate=0.1/s".
A no-op limiter (rate=0 for every provider) returns immediately from
`acquire(...)`, so unit tests behave exactly as they did before the
limiter was added.
"""
from __future__ import annotations
import pytest
from claude_code.rate_limiter import (
ALL_PROVIDERS,
ProviderConfig,
RateLimiter,
use_limiter,
)
@pytest.fixture(autouse=True)
def _hermetic_rate_limiter(tmp_path):
config = {p: ProviderConfig(rate_per_sec=0.0, burst=0.0) for p in ALL_PROVIDERS}
limiter = RateLimiter(config=config, state_dir=tmp_path)
with use_limiter(limiter):
yield

View file

@ -0,0 +1,201 @@
"""Unit tests for the shared `run_basic_messaging_cell` helper.
These tests mock `run_claude_models_parallel` so they exercise the
helper's branching (env-missing guard, per-model pass/fail/empty-text,
streaming wire check) without spawning the real CLI. The streaming
check is the regression we care about: a proxy that buffers the
upstream stream must turn the cell red, not green.
"""
from __future__ import annotations
import os
from typing import Any, Dict, List, Mapping, Optional, Sequence
import pytest
from claude_code import _basic_messaging
from claude_code._basic_messaging import (
MIN_STREAM_DELTA_EVENTS,
_count_stream_event_deltas,
run_basic_messaging_cell,
)
from claude_code.cli_driver import DriverResult
class _FakeResult:
"""Stand-in for the test's `compat_result` fixture.
Records every `set` / `add` payload so assertions can inspect what
the cell reported, in order, without needing the real
`pytest_runtest_logreport` plumbing from `conftest.py`.
"""
def __init__(self) -> None:
self.rows: List[Dict[str, Any]] = []
self.single: Optional[Dict[str, Any]] = None
def set(self, payload: Mapping[str, Any]) -> None:
self.single = dict(payload)
def add(self, payload: Mapping[str, Any]) -> None:
self.rows.append(dict(payload))
def _streamed_events(n_deltas: int = 5) -> List[Dict[str, Any]]:
"""Build a stream-json event list that *looks* streamed.
Includes `n_deltas` `stream_event` records (matching what
`--include-partial-messages` produces) plus the usual
`system`/`assistant`/`result` boilerplate the CLI always emits.
"""
events: List[Dict[str, Any]] = [{"type": "system", "subtype": "init"}]
for i in range(n_deltas):
events.append(
{
"type": "stream_event",
"event": {
"type": "content_block_delta",
"delta": {"type": "text_delta", "text": str(i)},
},
}
)
events.append(
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "1\n2\n3"}]},
}
)
events.append({"type": "result"})
return events
def _buffered_events() -> List[Dict[str, Any]]:
"""Event list a buffering proxy would produce: zero `stream_event`s."""
return [
{"type": "system", "subtype": "init"},
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "1\n2\n3"}]},
},
{"type": "result"},
]
def _install_fake_runner(monkeypatch, *, outcomes_by_model):
"""Patch `run_claude_models_parallel` to return canned outcomes.
Captures the kwargs the cell passed in so tests can assert on
`extra_args` (which is how the streaming variant opts into
`--include-partial-messages`).
"""
captured: Dict[str, Any] = {}
def fake(*, models, prompt, base_url, api_key, extra_args=None, **_kwargs):
captured["models"] = list(models)
captured["prompt"] = prompt
captured["base_url"] = base_url
captured["api_key"] = api_key
captured["extra_args"] = list(extra_args) if extra_args else []
return {model: outcomes_by_model[model] for model in models}
monkeypatch.setattr(_basic_messaging, "run_claude_models_parallel", fake)
return captured
@pytest.fixture(autouse=True)
def _proxy_env(monkeypatch):
monkeypatch.setenv("LITELLM_PROXY_BASE_URL", "http://localhost:4000")
monkeypatch.setenv("LITELLM_PROXY_API_KEY", "sk-test")
def test_count_stream_event_deltas_only_counts_records_with_event_payload():
events = [
{"type": "system"},
{"type": "stream_event", "event": {"type": "message_start"}},
{"type": "stream_event", "event": {"type": "content_block_delta"}},
{"type": "stream_event"},
{"type": "stream_event", "event": None},
{"type": "stream_event", "event": "not-a-dict"},
{"type": "assistant"},
{"type": "result"},
]
assert _count_stream_event_deltas(events) == 2
def test_verify_streaming_passes_when_proxy_streams(monkeypatch):
fake_result = _FakeResult()
model = "claude-haiku-4-5"
outcome = DriverResult(text="1\n2\n3", events=_streamed_events(n_deltas=5))
captured = _install_fake_runner(monkeypatch, outcomes_by_model={model: outcome})
run_basic_messaging_cell(
compat_result=fake_result,
models=[model],
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)
assert captured["extra_args"] == ["--include-partial-messages"]
assert fake_result.rows == [{"status": "pass"}]
def test_verify_streaming_fails_when_proxy_buffers(monkeypatch):
fake_result = _FakeResult()
model = "claude-haiku-4-5"
outcome = DriverResult(text="1\n2\n3", events=_buffered_events())
_install_fake_runner(monkeypatch, outcomes_by_model={model: outcome})
with pytest.raises(pytest.fail.Exception):
run_basic_messaging_cell(
compat_result=fake_result,
models=[model],
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)
assert len(fake_result.rows) == 1
row = fake_result.rows[0]
assert row["status"] == "fail"
assert "stream_event" in row["error"]
assert f"< {MIN_STREAM_DELTA_EVENTS}" in row["error"]
def test_non_streaming_variant_omits_partial_messages_flag(monkeypatch):
"""Default `verify_streaming=False` keeps the non-streaming wire identical."""
fake_result = _FakeResult()
model = "claude-haiku-4-5"
outcome = DriverResult(text="pong", events=_buffered_events())
captured = _install_fake_runner(monkeypatch, outcomes_by_model={model: outcome})
run_basic_messaging_cell(
compat_result=fake_result,
models=[model],
prompt="Reply with the single word 'pong' and nothing else.",
)
assert captured["extra_args"] == []
assert fake_result.rows == [{"status": "pass"}]
def test_verify_streaming_requires_all_models_to_stream(monkeypatch):
"""If any one tier buffers, the cell fails — same all-must-pass shape as
the non-streaming check."""
fake_result = _FakeResult()
outcomes = {
"claude-haiku-4-5": DriverResult(text="ok", events=_streamed_events(5)),
"claude-sonnet-4-6": DriverResult(text="ok", events=_buffered_events()),
"claude-opus-4-7": DriverResult(text="ok", events=_streamed_events(5)),
}
_install_fake_runner(monkeypatch, outcomes_by_model=outcomes)
with pytest.raises(pytest.fail.Exception):
run_basic_messaging_cell(
compat_result=fake_result,
models=list(outcomes.keys()),
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)
statuses = [row["status"] for row in fake_result.rows]
assert statuses == ["pass", "fail", "pass"]

View file

@ -0,0 +1,795 @@
"""Unit tests for the Claude Code CLI Driver.
These tests mock the subprocess so they run anywhere no network, no
`claude` install, no API keys. They cover the behavior contract:
argument assembly, environment overlay, stream-JSON parsing, exit-code
plumbing, and the structured failure modes (CLI not found, timeout).
"""
from __future__ import annotations
import json
import subprocess
from dataclasses import dataclass
from typing import List, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
DriverResult,
failure_diagnostic,
run_claude,
run_claude_models_parallel,
)
@dataclass
class _Completed:
returncode: int = 0
stdout: str = ""
stderr: str = ""
def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
captured = {}
def runner(cmd, env, capture_output, text, timeout, check, input=None):
captured["cmd"] = cmd
captured["env"] = env
captured["timeout"] = timeout
captured["input"] = input
return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
return runner, captured
def test_run_claude_assembles_command_correctly():
runner, captured = _make_runner(
stdout='{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}\n'
)
run_claude(
prompt="hello",
model="claude-haiku-4-5",
base_url="http://localhost:4000",
api_key="sk-test",
runner=runner,
)
cmd = captured["cmd"]
assert cmd[0] == "claude"
assert "--print" in cmd
assert "--output-format" in cmd
assert "stream-json" in cmd
assert "--model" in cmd
assert "claude-haiku-4-5" in cmd
# prompt is the last positional after the `--` end-of-options marker.
assert cmd[-2:] == ["--", "hello"]
def test_run_claude_places_extra_args_before_prompt():
"""`claude --print` expects the prompt as the final positional arg.
Flags appearing after the prompt are ignored or eaten by the prompt
parser (especially variadic flags like `--allowed-tools <tools...>`),
which silently broke the tool_use, vision, and web_search cells
before the fix. Pin the ordering: every flag (including
caller-supplied `extra_args`) must precede the `--` end-of-options
marker, which itself precedes the prompt.
"""
runner, captured = _make_runner(stdout="")
run_claude(
prompt="say hi",
model="claude-haiku-4-5",
base_url="http://localhost:4000",
api_key="sk-test",
extra_args=["--allowed-tools", "Bash"],
runner=runner,
)
cmd = captured["cmd"]
# Prompt is last, `--` immediately precedes it, and the caller's
# extra_args sit somewhere earlier in the command.
assert cmd[-2:] == ["--", "say hi"]
assert "--allowed-tools" in cmd
assert cmd.index("--allowed-tools") < cmd.index("--")
def test_run_claude_overlays_proxy_env():
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://proxy.example:4000",
api_key="sk-abc",
runner=runner,
)
env = captured["env"]
assert env["ANTHROPIC_BASE_URL"] == "http://proxy.example:4000"
assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-abc"
def test_run_claude_extra_env_is_added_to_subprocess_env():
"""Caller-supplied extra_env entries land on the subprocess env."""
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
extra_env={"MAX_THINKING_TOKENS": "4096"},
runner=runner,
)
assert captured["env"]["MAX_THINKING_TOKENS"] == "4096"
def test_run_claude_inherits_only_allowlisted_os_environ(monkeypatch):
"""Process-runtime vars (PATH) flow through; credentials don't.
The `claude` CLI is a Node binary installed dynamically from npm in
CI. If the package were ever compromised, inheriting the entire
parent environment would hand it every credential the surrounding
proxy job loads (AWS keys, Azure Foundry key, GitHub token, etc.).
Pin the contract: only the small allowlist of runtime vars is
inherited; everything else is dropped unless the caller passes it
explicitly via extra_env.
`HOME` is *not* on the allowlist anymore see the dedicated
isolated-HOME test below for the reason.
"""
monkeypatch.setenv("PATH", "/usr/bin:/usr/local/bin")
monkeypatch.setenv("HOME", "/home/runner")
monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "totally-secret")
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-proxy-only")
monkeypatch.setenv("AZURE_FOUNDRY_API_KEY", "azure-secret")
monkeypatch.setenv("VERTEXAI_CREDENTIALS", '{"private_key": "leak"}')
monkeypatch.setenv("GITHUB_TOKEN", "ghs_xxx")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
env = captured["env"]
assert env["PATH"] == "/usr/bin:/usr/local/bin"
assert "AWS_SECRET_ACCESS_KEY" not in env
assert "ANTHROPIC_API_KEY" not in env
assert "AZURE_FOUNDRY_API_KEY" not in env
assert "VERTEXAI_CREDENTIALS" not in env
assert "GITHUB_TOKEN" not in env
def test_run_claude_uses_isolated_per_invocation_home(monkeypatch, tmp_path):
"""`claude` subprocess never sees the runtime user's real $HOME.
The CLI needs *a* HOME (it caches per-session state under
`$HOME/.claude/projects/<sha>/`), but it has no business reading
the runtime user's real one. On the cron VM the runtime user is a
real interactive account with a populated home directory
(~/.config/gh/hosts.yml carrying a GitHub token, ~/.ssh/, etc.);
handing /home/mateo to a compromised npm package or to a
model-directed `Read` tool call during the PDF/vision cells
would let it exfiltrate those files. We hand the CLI a fresh
empty per-invocation tmpdir instead.
"""
monkeypatch.setenv("HOME", "/home/runner")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
env = captured["env"]
assert "HOME" in env, "claude CLI needs HOME to find ~/.claude session dir"
assert (
env["HOME"] != "/home/runner"
), "HOME must not leak the parent process's HOME to claude"
# The isolated HOME is a fresh tmpdir prefixed `claude-cli-home-`;
# see `_make_isolated_home` in cli_driver.py. It exists during the
# subprocess call and is removed afterwards (cleanup runs in a
# `finally`, so by the time this assertion runs the dir is gone —
# we only check the *prefix* of the path string we captured).
assert "claude-cli-home-" in env["HOME"]
def test_run_claude_isolated_home_is_distinct_per_invocation(monkeypatch):
"""Two consecutive calls get two different isolated HOMEs.
Reusing a single tmpdir across calls would defeat the isolation
in the parallel matrix run (a compromised CLI could plant a file
in HOME on one model's run and read it on the next). Pin: each
`run_claude` invocation gets its own freshly-created HOME.
"""
monkeypatch.setenv("HOME", "/home/runner")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
home_a = captured["env"]["HOME"]
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
home_b = captured["env"]["HOME"]
assert home_a != home_b
def test_run_claude_isolated_home_cleaned_up_after_run(monkeypatch):
"""The per-invocation HOME tmpdir is rm-rf'd when run_claude returns.
Without cleanup, a long matrix run would accumulate one tmpdir
per cell × per model × per CLI call (~75 dirs per cron run,
growing without bound across days).
"""
import os as _os
monkeypatch.setenv("HOME", "/home/runner")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
isolated_home = captured["env"]["HOME"]
assert not _os.path.exists(
isolated_home
), f"isolated HOME {isolated_home!r} should be removed after run_claude returns"
def test_run_claude_isolated_home_cleaned_up_on_subprocess_failure(monkeypatch):
"""Cleanup runs even when the CLI subprocess raises.
If the CLI is missing or times out, `run_claude` raises
`ClaudeCLIError` but the per-invocation HOME tmpdir must still
be removed (the `finally` clause), otherwise long failure-prone
runs leak tmpdirs.
"""
import os as _os
monkeypatch.setenv("HOME", "/home/runner")
captured: dict = {}
def runner(cmd, env, capture_output, text, timeout, check, input=None):
captured["env"] = env
raise subprocess.TimeoutExpired(cmd=cmd, timeout=timeout)
with pytest.raises(ClaudeCLIError):
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
isolated_home = captured["env"]["HOME"]
assert not _os.path.exists(
isolated_home
), f"isolated HOME {isolated_home!r} should be removed even on timeout"
def test_run_claude_extra_env_can_pass_through_otherwise_blocked_var(monkeypatch):
"""The allowlist applies to inherited os.environ; extra_env is the
sanctioned way for a test to opt-in to passing something extra."""
monkeypatch.setenv("ANTHROPIC_API_KEY", "from-os")
runner, captured = _make_runner(stdout="")
run_claude(
prompt="hi",
model="claude-opus-4-7",
base_url="http://localhost",
api_key="sk-abc",
extra_env={"ANTHROPIC_API_KEY": "from-arg"},
runner=runner,
)
assert captured["env"]["ANTHROPIC_API_KEY"] == "from-arg"
def test_run_claude_parses_stream_json_assistant_text():
events = [
{"type": "system", "session_id": "abc"},
{
"type": "assistant",
"message": {
"content": [
{"type": "text", "text": "Hello "},
{"type": "text", "text": "world"},
]
},
},
{"type": "result", "usage": {"input_tokens": 10, "output_tokens": 2}},
]
stdout = "\n".join(json.dumps(e) for e in events) + "\n"
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="claude-haiku-4-5",
base_url="http://localhost",
api_key="sk-abc",
runner=runner,
)
assert isinstance(result, DriverResult)
assert result.text == "Hello world"
assert len(result.events) == 3
assert result.usage == {"input_tokens": 10, "output_tokens": 2}
assert result.exit_code == 0
def test_run_claude_handles_string_message_content():
"""Some CLI versions emit `message.content` as a plain string."""
stdout = (
json.dumps({"type": "assistant", "message": {"content": "bare text"}}) + "\n"
)
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.text == "bare text"
def test_run_claude_skips_malformed_lines():
stdout = (
"not-json\n"
+ json.dumps(
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "x"}]},
}
)
+ "\n"
+ "{also-bad\n"
)
runner, _ = _make_runner(stdout=stdout)
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.text == "x"
assert len(result.events) == 1
def test_run_claude_propagates_nonzero_exit_code():
runner, _ = _make_runner(stdout="", returncode=2, stderr="auth failed")
result = run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
assert result.exit_code == 2
assert result.stderr == "auth failed"
assert result.text == ""
def test_run_claude_raises_on_missing_cli():
def runner(*args, **kwargs):
raise FileNotFoundError(2, "no such file", "claude")
with pytest.raises(ClaudeCLIError, match="claude CLI not found"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
def test_run_claude_raises_on_timeout():
def runner(*args, **kwargs):
raise subprocess.TimeoutExpired(cmd="claude", timeout=1)
with pytest.raises(ClaudeCLIError, match="timed out"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="k",
timeout=1,
runner=runner,
)
def test_run_claude_validates_required_params():
runner, _ = _make_runner()
with pytest.raises(ValueError, match="prompt"):
run_claude(
prompt="",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="stdin_input"):
run_claude(
prompt=None,
stdin_input="",
model="m",
base_url="http://x",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="model"):
run_claude(
prompt="hi",
model="",
base_url="http://x",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="base_url"):
run_claude(
prompt="hi",
model="m",
base_url="",
api_key="k",
runner=runner,
)
with pytest.raises(ValueError, match="api_key"):
run_claude(
prompt="hi",
model="m",
base_url="http://x",
api_key="",
runner=runner,
)
# ---------------------------------------------------------------------------
# failure_diagnostic
#
# Regression coverage for the bring-up incident where the proxy was started
# with the wrong config and tests reported only `claude CLI exited 1` while
# the actual 400 from LiteLLM was sitting in stdout. The helper must surface
# api_status, the assistant text (where API errors land), stderr, and the
# exit code together — and gracefully degrade when individual pieces are
# missing.
# ---------------------------------------------------------------------------
def test_failure_diagnostic_surfaces_api_error_text_from_stdout():
"""The CLI hides 4xx/5xx from the proxy in `assistant.message.content` text."""
api_error_text = (
'API Error: 400 {"error":{"message":"litellm.BadRequestError: '
"You passed in model=claude-haiku-4-5. There are no healthy "
'deployments..."}}'
)
result = DriverResult(
text=api_error_text,
events=[
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": api_error_text}]},
},
{
"type": "result",
"is_error": True,
"api_error_status": 400,
"result": api_error_text,
},
],
exit_code=1,
stderr="",
)
diag = failure_diagnostic(result)
assert "exit=1" in diag
assert "api_status=400" in diag
assert "There are no healthy deployments" in diag
def test_failure_diagnostic_falls_back_to_stderr_when_no_text():
result = DriverResult(text="", events=[], exit_code=2, stderr="boom\n")
diag = failure_diagnostic(result)
assert "exit=2" in diag
assert "stderr=boom" in diag
def test_failure_diagnostic_handles_completely_empty_result():
"""A run that produced literally nothing should still yield a useful string."""
result = DriverResult(text="", events=[], exit_code=137, stderr="")
diag = failure_diagnostic(result)
assert "exit=137" in diag
assert "no diagnostic output" in diag
def test_failure_diagnostic_truncates_long_text():
"""Don't let a 5MB HTML 502 page from a load balancer wreck the matrix JSON."""
huge = "x" * 5000
result = DriverResult(text=huge, events=[], exit_code=1, stderr="")
diag = failure_diagnostic(result, max_len=100)
assert "truncated" in diag
# Allow some slack for the prefix/suffix/separator characters.
assert len(diag) < 300
def test_failure_diagnostic_ignores_non_int_api_error_status():
"""The CLI sometimes emits api_error_status as a string; don't crash."""
result = DriverResult(
text="oops",
events=[{"type": "result", "api_error_status": "n/a"}],
exit_code=1,
stderr="",
)
diag = failure_diagnostic(result)
assert "api_status" not in diag
assert "text=oops" in diag
# ---------------------------------------------------------------------------
# run_claude_models_parallel
#
# The matrix runs three Claude tiers per cell, so the parallel helper has to
# (a) invoke `run_claude` once per model, (b) preserve each model's outcome
# separately, and (c) return errors as values rather than raising — callers
# need both the failed and the succeeded model results to report per-cell
# rows accurately.
# ---------------------------------------------------------------------------
def test_run_claude_models_parallel_returns_one_result_per_model():
"""Each model gets its own DriverResult keyed under the helper's dict."""
seen_models: List[str] = []
def runner(cmd, env, capture_output, text, timeout, check, input=None):
# The model id is two slots after `--model` in the assembled command.
idx = cmd.index("--model")
model = cmd[idx + 1]
seen_models.append(model)
return _Completed(
returncode=0,
stdout=json.dumps(
{
"type": "assistant",
"message": {
"content": [{"type": "text", "text": f"reply-{model}"}]
},
}
)
+ "\n",
)
outcomes = run_claude_models_parallel(
models=["a", "b", "c"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
assert set(outcomes.keys()) == {"a", "b", "c"}
for model in ("a", "b", "c"):
result = outcomes[model]
assert isinstance(result, DriverResult)
assert result.text == f"reply-{model}"
assert result.exit_code == 0
assert sorted(seen_models) == ["a", "b", "c"]
def test_run_claude_models_parallel_returns_errors_as_values():
"""A model whose CLI is missing surfaces as a ClaudeCLIError, not a raise."""
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
if model == "boom":
raise FileNotFoundError(2, "no such file", "claude")
return _Completed(
returncode=0,
stdout=json.dumps(
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "ok"}]},
}
)
+ "\n",
)
outcomes = run_claude_models_parallel(
models=["ok-model", "boom"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
assert isinstance(outcomes["ok-model"], DriverResult)
assert outcomes["ok-model"].text == "ok"
assert isinstance(outcomes["boom"], ClaudeCLIError)
assert "claude CLI not found" in str(outcomes["boom"])
def test_run_claude_models_parallel_preserves_nonzero_exit_codes():
"""Mixed success/failure on exit code should not collapse into one verdict."""
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
if model == "fail":
return _Completed(returncode=2, stdout="", stderr="auth failed")
return _Completed(
returncode=0,
stdout=json.dumps(
{
"type": "assistant",
"message": {"content": [{"type": "text", "text": "ok"}]},
}
)
+ "\n",
)
outcomes = run_claude_models_parallel(
models=["ok-model", "fail"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
assert outcomes["ok-model"].exit_code == 0
assert outcomes["fail"].exit_code == 2
assert outcomes["fail"].stderr == "auth failed"
def test_run_claude_models_parallel_rejects_empty_models():
with pytest.raises(ValueError, match="non-empty"):
run_claude_models_parallel(
models=[],
prompt="hi",
base_url="http://x",
api_key="k",
)
def test_run_claude_models_parallel_stamps_duration_on_each_result():
"""Each DriverResult carries the per-model wall time so callers can
attribute slow cells without re-timing the work themselves.
The fake runner sleeps for very different durations per model so
we can prove each result is timing its own work (not the batch
wall time). We use generous absolute bounds because thread-pool
scheduling on a loaded CI box adds noise on the order of tens of
milliseconds.
"""
import time
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
time.sleep(0.05 if model == "fast" else 0.40)
return _Completed(returncode=0, stdout="")
outcomes = run_claude_models_parallel(
models=["fast", "slow"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
fast_ms = outcomes["fast"].duration_ms
slow_ms = outcomes["slow"].duration_ms
assert fast_ms is not None and slow_ms is not None
# 50ms sleep ⇒ ~50250ms after scheduling overhead; 400ms sleep ⇒
# 400700ms. We just need the two distributions to be non-overlapping
# so we know each row's duration is its own work, not the batch's.
assert fast_ms < 300, fast_ms
assert slow_ms >= 350, slow_ms
assert slow_ms > fast_ms
def test_run_claude_models_parallel_breakdown_logs_to_stderr(capsys):
"""The breakdown helper must emit a per-model timing block so users
can answer "why didn't parallel help?" without re-instrumenting."""
def runner(cmd, env, capture_output, text, timeout, check, input=None):
return _Completed(returncode=0, stdout="")
run_claude_models_parallel(
models=["model-x", "model-y"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
captured = capsys.readouterr()
assert "[parallel] per-model wall time:" in captured.err
assert "model-x" in captured.err
assert "model-y" in captured.err
assert "speedup=" in captured.err
assert "slowest=" in captured.err
def test_run_claude_models_parallel_breakdown_marks_cli_errors(capsys):
"""When a model raises ClaudeCLIError, the breakdown should still
show its row tagged as `cli-error` rather than crashing or omitting it."""
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
if cmd[idx + 1] == "boom":
raise FileNotFoundError(2, "no such file", "claude")
return _Completed(returncode=0, stdout="")
run_claude_models_parallel(
models=["ok-model", "boom"],
prompt="hi",
base_url="http://x",
api_key="k",
runner=runner,
)
captured = capsys.readouterr()
assert "ok-model" in captured.err
assert "boom" in captured.err
assert "cli-error" in captured.err
def test_run_claude_models_parallel_forwards_extra_args_and_env():
"""Shared kwargs must reach every per-model invocation unchanged."""
captured_envs: List[dict] = []
captured_cmds: List[List[str]] = []
def runner(cmd, env, capture_output, text, timeout, check, input=None):
captured_envs.append(env)
captured_cmds.append(cmd)
return _Completed(returncode=0, stdout="")
run_claude_models_parallel(
models=["a", "b"],
prompt="hi",
base_url="http://x",
api_key="k",
extra_env={"MAX_THINKING_TOKENS": "4096"},
extra_args=["--allowed-tools", "Bash"],
runner=runner,
)
assert all(env["MAX_THINKING_TOKENS"] == "4096" for env in captured_envs)
for cmd in captured_cmds:
assert "--allowed-tools" in cmd
assert "Bash" in cmd
def test_failure_diagnostic_uses_last_result_event_status():
"""If multiple `result` events appear, the most recent status wins."""
result = DriverResult(
text="",
events=[
{"type": "result", "api_error_status": 500},
{"type": "assistant", "message": {"content": []}},
{"type": "result", "api_error_status": 429},
],
exit_code=1,
stderr="",
)
diag = failure_diagnostic(result)
assert "api_status=429" in diag
assert "500" not in diag

View file

@ -0,0 +1,138 @@
"""Tests for the `compat_result` fixture's tagged-union validation.
The conftest's `pytest_runtest_makereport` hook is exercised end-to-end by
the matrix-builder golden-file tests (which consume a results.json that
the harness would produce). Here we just test the input-validation
contract on `CompatResult.set()`.
"""
from __future__ import annotations
import pytest
from claude_code.conftest import CompatResult
def test_set_pass_is_accepted():
r = CompatResult()
r.set({"status": "pass"})
assert r.value == {"status": "pass"}
def test_set_fail_requires_error():
r = CompatResult()
with pytest.raises(ValueError, match="requires 'error'"):
r.set({"status": "fail"})
def test_set_fail_with_error_is_accepted():
r = CompatResult()
r.set({"status": "fail", "error": "boom"})
assert r.value == {"status": "fail", "error": "boom"}
def test_set_not_applicable_requires_reason():
r = CompatResult()
with pytest.raises(ValueError, match="requires 'reason'"):
r.set({"status": "not_applicable"})
def test_set_not_applicable_with_reason_is_accepted():
r = CompatResult()
r.set({"status": "not_applicable", "reason": "Bedrock has no /thinking"})
assert r.value == {"status": "not_applicable", "reason": "Bedrock has no /thinking"}
def test_set_not_tested_is_accepted():
r = CompatResult()
r.set({"status": "not_tested"})
assert r.value == {"status": "not_tested"}
def test_set_rejects_unknown_status():
r = CompatResult()
with pytest.raises(ValueError, match="status must be one of"):
r.set({"status": "maybe"})
def test_set_rejects_non_dict():
r = CompatResult()
with pytest.raises(TypeError):
r.set("pass") # type: ignore[arg-type]
def test_set_copies_input():
"""Mutating the dict after set() must not change the stored value."""
r = CompatResult()
payload = {"status": "fail", "error": "x"}
r.set(payload)
payload["error"] = "mutated"
assert r.value["error"] == "x"
# ---------------------------------------------------------------------------
# add() / collected()
#
# When a single test exercises three Claude tiers in parallel, each tier
# needs its own row in the results artifact so the matrix builder can
# apply its "all three must pass" aggregation. `add()` is the per-tier
# recorder; `collected()` is what the conftest hook reads.
# ---------------------------------------------------------------------------
def test_add_appends_each_call_to_values():
r = CompatResult()
r.add({"status": "pass"})
r.add({"status": "fail", "error": "bad"})
assert r.values == [
{"status": "pass"},
{"status": "fail", "error": "bad"},
]
def test_add_validates_like_set():
"""The add() and set() validators are the same; both must reject bad payloads."""
r = CompatResult()
with pytest.raises(ValueError, match="requires 'error'"):
r.add({"status": "fail"})
with pytest.raises(ValueError, match="requires 'reason'"):
r.add({"status": "not_applicable"})
with pytest.raises(ValueError, match="status must be one of"):
r.add({"status": "maybe"})
with pytest.raises(TypeError):
r.add("pass") # type: ignore[arg-type]
def test_add_copies_input():
"""Same defensive copy contract as set()."""
r = CompatResult()
payload = {"status": "fail", "error": "x"}
r.add(payload)
payload["error"] = "mutated"
assert r.values[0]["error"] == "x"
def test_collected_returns_values_when_added():
r = CompatResult()
r.add({"status": "pass"})
r.add({"status": "pass"})
assert r.collected() == [{"status": "pass"}, {"status": "pass"}]
def test_collected_returns_single_value_when_only_set_called():
"""Legacy single-result tests should still surface their one outcome."""
r = CompatResult()
r.set({"status": "pass"})
assert r.collected() == [{"status": "pass"}]
def test_collected_prefers_added_values_over_set_value():
"""If both are populated, the per-tier list wins — that's the multi-model shape."""
r = CompatResult()
r.set({"status": "pass"})
r.add({"status": "fail", "error": "tier-2 broke"})
assert r.collected() == [{"status": "fail", "error": "tier-2 broke"}]
def test_collected_returns_empty_when_nothing_reported():
assert CompatResult().collected() == []

View file

@ -0,0 +1,329 @@
"""Unit tests for the cross-process token-bucket rate limiter.
The tests cover three layers:
1. Provider inference from model alias the matrix-column mapping the
live tests rely on (`-bedrock-converse` vs `-bedrock-invoke` vs
`-azure` vs `-vertex` vs bare = anthropic).
2. Config parsing env-var precedence, fallback to default, malformed
input handling, burst override semantics. These run against
`os.environ`-shaped dicts so we don't have to monkeypatch globals.
3. Token-bucket behavior enforcing rate, accumulating burst, never
over-spending across a fake clock. Filesystem state is exercised
with a real `tmp_path` because the persistence is the whole point;
the only injected seam is `clock` (and `sleep`, so tests don't
actually wait on wall time).
The cross-process flock semantics are exercised indirectly: every
test creates a fresh `RateLimiter` rooted at `tmp_path`, so the same
file lock that protects production is exercised here too. We don't
fork to test multi-process behavior in this file because pytest
fixtures + xdist already do that for the integration suite.
"""
from __future__ import annotations
import json
import time
from pathlib import Path
from typing import List
import pytest
from claude_code.rate_limiter import (
ALL_PROVIDERS,
BURST_ENV,
DEFAULT_RATE,
PROVIDER_ANTHROPIC,
PROVIDER_AZURE,
PROVIDER_BEDROCK_CONVERSE,
PROVIDER_BEDROCK_INVOKE,
PROVIDER_VERTEX_AI,
ProviderConfig,
RateLimiter,
get_default_limiter,
infer_provider,
load_config,
reset_default_limiter,
use_limiter,
)
# ---------------------------------------------------------------------------
# Provider inference
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"model, expected",
[
("claude-haiku-4-5", PROVIDER_ANTHROPIC),
("claude-sonnet-4-6", PROVIDER_ANTHROPIC),
("claude-opus-4-7", PROVIDER_ANTHROPIC),
("claude-haiku-4-5-azure", PROVIDER_AZURE),
("claude-sonnet-4-6-azure", PROVIDER_AZURE),
("claude-opus-4-7-vertex", PROVIDER_VERTEX_AI),
("claude-haiku-4-5-bedrock-converse", PROVIDER_BEDROCK_CONVERSE),
("claude-haiku-4-5-bedrock-invoke", PROVIDER_BEDROCK_INVOKE),
],
)
def test_infer_provider_maps_alias_suffix_to_column(model, expected):
assert infer_provider(model) == expected
def test_infer_provider_bedrock_converse_beats_bedrock_invoke_lookup_order():
"""Both bedrock suffixes contain `bedrock`; the more-specific suffix wins."""
assert infer_provider("claude-foo-bedrock-converse") == PROVIDER_BEDROCK_CONVERSE
assert infer_provider("claude-foo-bedrock-invoke") == PROVIDER_BEDROCK_INVOKE
def test_infer_provider_rejects_empty_string():
with pytest.raises(ValueError, match="non-empty"):
infer_provider("")
def test_infer_provider_is_case_insensitive():
"""Aliases in the proxy config sometimes drift between cases; we
should still route them to the right column."""
assert infer_provider("CLAUDE-OPUS-4-7-AZURE") == PROVIDER_AZURE
# ---------------------------------------------------------------------------
# Config parsing
# ---------------------------------------------------------------------------
def test_load_config_uses_default_rate_when_env_absent():
cfg = load_config(env={})
for provider in ALL_PROVIDERS:
assert cfg[provider].rate_per_sec == DEFAULT_RATE
assert cfg[provider].burst == DEFAULT_RATE
def test_load_config_reads_per_provider_rate():
cfg = load_config(
env={
"LITELLM_COMPAT_RATE_ANTHROPIC": "10",
"LITELLM_COMPAT_RATE_AZURE": "0.5",
}
)
assert cfg[PROVIDER_ANTHROPIC].rate_per_sec == 10.0
assert cfg[PROVIDER_AZURE].rate_per_sec == 0.5
assert cfg[PROVIDER_VERTEX_AI].rate_per_sec == DEFAULT_RATE
def test_load_config_zero_rate_disables_provider():
cfg = load_config(env={"LITELLM_COMPAT_RATE_BEDROCK_INVOKE": "0"})
assert cfg[PROVIDER_BEDROCK_INVOKE].enabled is False
def test_load_config_burst_override_applies_to_every_provider():
cfg = load_config(
env={
"LITELLM_COMPAT_RATE_ANTHROPIC": "5",
BURST_ENV: "20",
}
)
for provider in ALL_PROVIDERS:
assert cfg[provider].burst == 20.0
def test_load_config_falls_back_on_malformed_value():
cfg = load_config(env={"LITELLM_COMPAT_RATE_ANTHROPIC": "not-a-number"})
assert cfg[PROVIDER_ANTHROPIC].rate_per_sec == DEFAULT_RATE
def test_load_config_burst_floors_at_one_when_rate_is_low():
"""A 0.5/s rate with no burst override must still allow at least
one immediate request otherwise the very first call would block."""
cfg = load_config(env={"LITELLM_COMPAT_RATE_ANTHROPIC": "0.5"})
assert cfg[PROVIDER_ANTHROPIC].burst == 1.0
# ---------------------------------------------------------------------------
# Token bucket
# ---------------------------------------------------------------------------
@pytest.fixture
def fake_clock():
"""A controllable monotonic clock + sleep for the limiter under test.
Tests advance `clock.now` to simulate elapsed wall time. `sleep`
adds the requested duration to `clock.now` instead of actually
sleeping, so a "wait 200ms" code path runs in microseconds and
is deterministic.
"""
class Clock:
def __init__(self):
self.now = 1_000.0
self.sleeps: List[float] = []
def __call__(self):
return self.now
def sleep(self, seconds: float) -> None:
self.sleeps.append(seconds)
self.now += seconds
return Clock()
def _make_limiter(tmp_path: Path, fake_clock, *, rate=10.0, burst=None):
cfg = {
p: ProviderConfig(rate_per_sec=rate, burst=burst if burst is not None else rate)
for p in ALL_PROVIDERS
}
return RateLimiter(
config=cfg,
state_dir=tmp_path,
clock=fake_clock,
sleep=fake_clock.sleep,
)
def test_acquire_first_call_does_not_wait(tmp_path, fake_clock):
"""A freshly-initialized bucket starts full; the first acquire is free."""
limiter = _make_limiter(tmp_path, fake_clock, rate=10.0, burst=10.0)
waited = limiter.acquire(PROVIDER_ANTHROPIC)
assert waited == 0.0
assert fake_clock.sleeps == []
def test_acquire_disabled_provider_returns_immediately(tmp_path, fake_clock):
"""rate=0 ⇒ no throttling, even if every other provider is throttled."""
cfg = {p: ProviderConfig(rate_per_sec=0.0, burst=0.0) for p in ALL_PROVIDERS}
limiter = RateLimiter(
config=cfg, state_dir=tmp_path, clock=fake_clock, sleep=fake_clock.sleep
)
for _ in range(100):
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
assert fake_clock.sleeps == []
def test_acquire_burns_through_burst_then_throttles(tmp_path, fake_clock):
"""`burst` immediate requests succeed; the next one waits 1/rate seconds."""
limiter = _make_limiter(tmp_path, fake_clock, rate=2.0, burst=3.0)
for _ in range(3):
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
# Bucket is empty; next call must sleep ~0.5s to earn one token at 2/s.
waited = limiter.acquire(PROVIDER_ANTHROPIC)
assert waited == pytest.approx(0.5, abs=0.01)
def test_acquire_refills_with_elapsed_time(tmp_path, fake_clock):
"""Advancing the clock between calls credits tokens at the configured rate."""
limiter = _make_limiter(tmp_path, fake_clock, rate=4.0, burst=1.0)
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0 # consumes the 1-token burst
fake_clock.now += 0.25 # 0.25s × 4/s = 1 token earned
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
def test_acquire_caps_refill_at_burst(tmp_path, fake_clock):
"""A long quiet period must not let the bucket grow past `burst`."""
limiter = _make_limiter(tmp_path, fake_clock, rate=10.0, burst=2.0)
fake_clock.now += 1_000 # would earn 10_000 tokens uncapped
# Only `burst` (=2) immediate calls should succeed before throttling.
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
waited = limiter.acquire(PROVIDER_ANTHROPIC)
assert waited > 0
def test_acquire_independent_buckets_per_provider(tmp_path, fake_clock):
"""Anthropic exhaustion must not throttle Azure (each column has its own bucket)."""
limiter = _make_limiter(tmp_path, fake_clock, rate=2.0, burst=1.0)
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
# Anthropic bucket is now empty; Azure is untouched.
assert limiter.acquire(PROVIDER_AZURE) == 0.0
def test_acquire_persists_state_across_limiter_instances(tmp_path):
"""A fresh RateLimiter must read the on-disk state, not start fresh.
This is the property that makes the limiter cross-process: an
xdist worker created mid-run sees the credit consumed by other
workers, instead of getting its own private bucket.
"""
cfg = {p: ProviderConfig(rate_per_sec=10.0, burst=2.0) for p in ALL_PROVIDERS}
state = {"now": 1_000.0, "sleeps": []}
def clock():
return state["now"]
def sleep(seconds):
state["sleeps"].append(seconds)
state["now"] += seconds
first = RateLimiter(config=cfg, state_dir=tmp_path, clock=clock, sleep=sleep)
first.acquire(PROVIDER_ANTHROPIC)
first.acquire(PROVIDER_ANTHROPIC)
# bucket is now empty
second = RateLimiter(config=cfg, state_dir=tmp_path, clock=clock, sleep=sleep)
waited = second.acquire(PROVIDER_ANTHROPIC)
assert waited > 0 # had to wait, didn't see a fresh full bucket
def test_acquire_recovers_from_corrupt_state_file(tmp_path, fake_clock):
"""A truncated/garbage state file must not crash the test session."""
state_file = tmp_path / f"{PROVIDER_ANTHROPIC}.json"
state_file.write_text("not-json {{")
limiter = _make_limiter(tmp_path, fake_clock, rate=5.0, burst=5.0)
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
def test_acquire_handles_clock_going_backward(tmp_path, fake_clock):
"""Across a host suspend/resume the monotonic clock can briefly
go backward; we must not interpret that as removing tokens."""
limiter = _make_limiter(tmp_path, fake_clock, rate=1.0, burst=2.0)
limiter.acquire(PROVIDER_ANTHROPIC)
fake_clock.now -= 10 # clock moved backward
# Bucket should still have ~1 token left from the burst, not -9.
assert limiter.acquire(PROVIDER_ANTHROPIC) == 0.0
# ---------------------------------------------------------------------------
# Process-default singleton
# ---------------------------------------------------------------------------
def test_use_limiter_swaps_default_for_block(tmp_path):
sentinel_cfg = {
p: ProviderConfig(rate_per_sec=0.0, burst=0.0) for p in ALL_PROVIDERS
}
sentinel = RateLimiter(config=sentinel_cfg, state_dir=tmp_path)
reset_default_limiter()
try:
with use_limiter(sentinel):
assert get_default_limiter() is sentinel
# After the context exits, the default goes back to whatever it
# was — in this test that's "rebuilt on next access" because we
# called reset_default_limiter() above.
assert get_default_limiter() is not sentinel
finally:
reset_default_limiter()
# ---------------------------------------------------------------------------
# Persistence shape
# ---------------------------------------------------------------------------
def test_state_file_is_json_after_acquire(tmp_path, fake_clock):
limiter = _make_limiter(tmp_path, fake_clock, rate=5.0, burst=5.0)
limiter.acquire(PROVIDER_ANTHROPIC)
state_file = tmp_path / f"{PROVIDER_ANTHROPIC}.json"
payload = json.loads(state_file.read_text())
assert "tokens" in payload
assert "last_refill" in payload
assert payload["tokens"] == pytest.approx(4.0)

View file

@ -0,0 +1,147 @@
"""Pin tests for the `Bash`-using compat cells.
Every cell that passes `--allowed-tools Bash` to the `claude` CLI is
giving a model-controlled response the ability to run host commands.
On the PR-gate CircleCI machine executor, those commands have access
to the Docker socket and can read `docker inspect compat-proxy` to
recover the provider credentials living inside the proxy container.
To narrow that surface, every Bash-using cell must:
1. Restrict the allow rule to the *exact* command `Bash(echo pong)` so
a compromised provider response cannot turn `Bash` into arbitrary
host execution by emitting a `tool_use` with a different command.
2. Pair it with `--permission-mode dontAsk` so anything not matching
an allow rule is auto-denied instead of prompting (which would
abort the CLI in headless mode, but auto-denial is the explicit
contract).
These restrictions are enforced by the `claude` CLI, not by the
model see https://code.claude.com/docs/en/permissions for the
permission-rule precedence (`deny` `ask` `allow`).
This test scans every cell under the three Bash-using feature
directories (`tool_use`, `tool_use_streaming`, `thinking_with_tool_use`)
and pins both requirements so a future test refactor cannot silently
revert any cell to the broad `Bash` allow that was originally
flagged by Veria.
"""
from __future__ import annotations
from pathlib import Path
from typing import Iterable
import pytest
REPO_ROOT = Path(__file__).resolve().parents[4]
CLAUDE_CODE_DIR = REPO_ROOT / "tests" / "e2e" / "claude_code"
# Feature directories whose cells drive the `Bash` built-in tool. Add
# new entries here when a new Bash-using feature is added; the test
# fails loudly for any unhandled directory so we never miss one by
# silent omission.
BASH_FEATURE_DIRS = (
"tool_use",
"tool_use_streaming",
"thinking_with_tool_use",
)
def _bash_cells() -> Iterable[Path]:
for feature in BASH_FEATURE_DIRS:
feature_dir = CLAUDE_CODE_DIR / feature
assert feature_dir.is_dir(), (
f"{feature_dir} is missing — BASH_FEATURE_DIRS is out of sync "
f"with the layout under tests/e2e/claude_code/."
)
for path in sorted(feature_dir.glob("test_*.py")):
yield path
def _has_bare_bash_token(text: str) -> bool:
"""Return True if `text` contains a `"Bash"` token outside the
`"Bash(echo pong)"` allow rule.
Extracted as a pure helper so the negative path can be unit-tested
directly. Without it, the previous structure of this assertion was
`'"Bash"' not in text or '"Bash(echo pong)"' in text`, which
short-circuits to True any time the allow rule is present and lets
a stray bare `"Bash"` slip through the security pin undetected.
"""
return '"Bash"' in text.replace('"Bash(echo pong)"', "")
@pytest.mark.parametrize(
"cell", list(_bash_cells()), ids=lambda p: str(p.relative_to(REPO_ROOT))
)
def test_bash_allow_rule_is_pinned_to_exact_echo_pong(cell: Path) -> None:
"""The cell must pass `Bash(echo pong)` as the allow rule, not the
unrestricted `Bash` value that was originally flagged."""
text = cell.read_text()
assert '"Bash(echo pong)"' in text, (
f"{cell.relative_to(REPO_ROOT)} must restrict `--allowed-tools` to "
f'`Bash(echo pong)` (exact-match pattern). Unrestricted `"Bash"` '
f"grants arbitrary host command execution to model-controlled "
f"tool_use blocks, which can read `docker inspect compat-proxy` "
f"to exfiltrate provider credentials from the proxy container."
)
# The only place `"Bash"` (the bare token, surrounded by quotes
# exactly as it would appear in `--allowed-tools` lists) is allowed
# to appear is *inside* the exact-match `"Bash(echo pong)"` rule.
# `_has_bare_bash_token` keeps that scan independent of the first
# assertion — otherwise `'"Bash"' not in text or '"Bash(echo pong)"'
# in text` short-circuits to True and lets a stray bare `"Bash"`
# slip through silently.
assert not _has_bare_bash_token(text), (
f"{cell.relative_to(REPO_ROOT)} still references the unrestricted "
f'`"Bash"` value outside the `"Bash(echo pong)"` allow rule — '
f"sweep it out before merging."
)
def test_has_bare_bash_token_flags_unrestricted_value():
"""A file that allows the bare `"Bash"` token alongside the
exact-match rule must be flagged. Without this guard the security
pin reverts to the dead-code `or` it had originally, which let
arbitrary host commands through under the noise of a passing test.
"""
text = '--allowed-tools "Bash" "Bash(echo pong)"'
assert _has_bare_bash_token(text)
def test_has_bare_bash_token_accepts_only_exact_match():
"""The standard pattern — only the exact-match allow rule, no bare
`"Bash"` must be accepted. This is the shape every Bash-using
cell in the suite is required to take.
"""
text = '--allowed-tools "Bash(echo pong)" --permission-mode "dontAsk"'
assert not _has_bare_bash_token(text)
def test_has_bare_bash_token_ignores_unrelated_substrings():
"""`Bash(echo pong)` is the only allowed shape; substrings like
`BashTool` or `Bashing` are unrelated identifiers and must not be
confused with the bare `"Bash"` token (i.e. the exact quoted
string `"Bash"`)."""
text = "BashTool helper used by the bashing harness"
assert not _has_bare_bash_token(text)
@pytest.mark.parametrize(
"cell", list(_bash_cells()), ids=lambda p: str(p.relative_to(REPO_ROOT))
)
def test_bash_cell_uses_dontask_permission_mode(cell: Path) -> None:
"""The cell must pair the allow rule with `--permission-mode dontAsk`
so tool calls that don't match the allow rule are auto-denied (as
opposed to defaulting to "ask", which in headless mode would
succeed without ever surfacing the security issue)."""
text = cell.read_text()
assert '"--permission-mode"' in text and '"dontAsk"' in text, (
f"{cell.relative_to(REPO_ROOT)} must pass `--permission-mode dontAsk` "
f"alongside the `Bash(echo pong)` allow rule. Without dontAsk, "
f"commands outside the allow rule fall back to the default ask-"
f"mode behavior, which in `--print` (headless) mode is non-"
f"interactive — defeating the explicit-allow contract."
)

View file

@ -0,0 +1,164 @@
"""Unit tests for the Claude Code PR-Gate Version Resolver.
The resolver picks the newest `@anthropic-ai/claude-code` version whose
publish timestamp is at least 3 days old. The 3-day window is a security
review buffer: a malicious or broken Claude Code release that slipped
through the npm publish process gets at least 72 hours to be detected
before it can land in the LiteLLM PR gate.
The unit tests inject npm metadata directly (no network) and a fixed
`as_of` clock (no real time), so they run anywhere and never flake on
the wall clock or registry availability.
"""
from __future__ import annotations
from datetime import datetime, timedelta, timezone
import pytest
from claude_code.pr_gate_version_resolver import (
NoEligibleVersionError,
resolve_pr_gate_version,
)
def _t(iso: str) -> str:
"""Helper for readable ISO-8601 publish timestamps in fixtures."""
return iso
# A clock fixed at a moment well after every fixture publish time below.
NOW = datetime(2026, 4, 25, 12, 0, 0, tzinfo=timezone.utc)
def _metadata_with_times(times: dict) -> dict:
"""Shape an npm `packument`-like dict with the `time` field populated.
The npm registry response includes `time.created` / `time.modified`
keys alongside per-version timestamps; the resolver must skip those.
"""
return {
"name": "@anthropic-ai/claude-code",
"time": {
"created": _t("2024-01-01T00:00:00.000Z"),
"modified": _t("2026-04-25T00:00:00.000Z"),
**times,
},
}
def test_picks_newest_version_at_least_three_days_old():
metadata = _metadata_with_times(
{
"2.1.118": _t("2026-04-15T10:00:00.000Z"),
"2.1.119": _t("2026-04-21T10:00:00.000Z"), # 4d 2h old
"2.1.120": _t("2026-04-23T10:00:00.000Z"), # 2d 2h old — too new
"2.1.121": _t("2026-04-25T11:00:00.000Z"), # 1h old — too new
}
)
assert resolve_pr_gate_version(metadata=metadata, as_of=NOW) == "2.1.119"
def test_skips_created_and_modified_meta_keys():
"""`time` contains `created` / `modified` non-version entries — must be ignored."""
metadata = {
"name": "@anthropic-ai/claude-code",
"time": {
"created": _t("2024-01-01T00:00:00.000Z"),
"modified": _t("2026-04-25T00:00:00.000Z"),
"2.0.0": _t("2026-04-10T00:00:00.000Z"),
},
}
assert resolve_pr_gate_version(metadata=metadata, as_of=NOW) == "2.0.0"
def test_min_age_boundary_is_inclusive():
"""A version published exactly 3 days ago is eligible (>= cutoff)."""
three_days_ago = NOW - timedelta(days=3)
metadata = _metadata_with_times(
{
"2.1.0": three_days_ago.isoformat().replace("+00:00", "Z"),
}
)
assert resolve_pr_gate_version(metadata=metadata, as_of=NOW) == "2.1.0"
def test_raises_when_every_version_is_too_new():
metadata = _metadata_with_times(
{
"2.1.121": _t("2026-04-25T08:00:00.000Z"), # 4h old
"2.1.120": _t("2026-04-24T10:00:00.000Z"), # ~26h old
}
)
with pytest.raises(NoEligibleVersionError):
resolve_pr_gate_version(metadata=metadata, as_of=NOW)
def test_raises_when_metadata_has_no_versions():
metadata = {"name": "@anthropic-ai/claude-code", "time": {}}
with pytest.raises(NoEligibleVersionError):
resolve_pr_gate_version(metadata=metadata, as_of=NOW)
def test_picks_latest_publish_time_not_largest_semver():
"""If a patch is published to an old major after a newer release,
"newest" is by publish time, not semver string ordering."""
metadata = _metadata_with_times(
{
"1.9.99": _t("2026-04-22T10:00:00.000Z"), # patched recently — wins
"2.0.0": _t("2026-03-01T10:00:00.000Z"), # older publish
}
)
assert resolve_pr_gate_version(metadata=metadata, as_of=NOW) == "1.9.99"
def test_uses_custom_min_age():
metadata = _metadata_with_times(
{
"1.0.0": _t("2026-04-23T10:00:00.000Z"), # 2d 2h old
"0.9.0": _t("2026-04-10T10:00:00.000Z"), # 15d old
}
)
# min_age = 5 days disqualifies 1.0.0
out = resolve_pr_gate_version(
metadata=metadata, as_of=NOW, min_age=timedelta(days=5)
)
assert out == "0.9.0"
def test_excludes_prerelease_versions():
"""Pre-release tags (1.0.0-alpha.1, 2.0.0-rc.1, etc.) must never win,
even if their publish timestamp is the newest eligible one."""
metadata = _metadata_with_times(
{
"2.1.119": _t("2026-04-21T10:00:00.000Z"), # stable, 4d old
"2.2.0-alpha.1": _t("2026-04-22T10:00:00.000Z"), # newer publish
"2.2.0-rc.1": _t("2026-04-22T11:00:00.000Z"), # newest publish
"3.0.0-beta": _t("2026-04-22T12:00:00.000Z"), # newest publish
}
)
assert resolve_pr_gate_version(metadata=metadata, as_of=NOW) == "2.1.119"
def test_raises_when_only_prereleases_are_eligible():
metadata = _metadata_with_times(
{
"2.2.0-alpha.1": _t("2026-04-22T10:00:00.000Z"),
"2.2.0-rc.1": _t("2026-04-22T11:00:00.000Z"),
}
)
with pytest.raises(NoEligibleVersionError):
resolve_pr_gate_version(metadata=metadata, as_of=NOW)
def test_resolver_uses_fetcher_when_metadata_not_provided():
captured = {}
def fake_fetch(package_name: str) -> dict:
captured["package"] = package_name
return _metadata_with_times({"3.0.0": _t("2026-04-10T10:00:00.000Z")})
out = resolve_pr_gate_version(as_of=NOW, fetcher=fake_fetch)
assert out == "3.0.0"
assert captured["package"] == "@anthropic-ai/claude-code"

View file

@ -0,0 +1,95 @@
"""Pin: the cron `pytest` invocation must run under `env -i`.
The systemd service `litellm-compat-matrix.service` loads provider
credentials (`ANTHROPIC_API_KEY`, `AWS_BEARER_TOKEN_BEDROCK`,
`AZURE_FOUNDRY_API_KEY`, `VERTEXAI_*`) and the agent-shin GitHub token
(`AGENT_SHIN_GITHUB_TOKEN`) into `run_daily.sh`'s environment from
`/etc/litellm-compat-matrix.env`. Pytest only needs to talk to the
loopback proxy at `127.0.0.1:${PROXY_PORT}` and has no legitimate reason
to see provider creds in its own `os.environ`. Leaving them in would
let a test under `tests/e2e/claude_code/` read them via `os.environ` and
exfiltrate them, and would also let a model-directed `Read` tool call
during a PDF/vision cell reach `/proc/<pytest-pid>/environ`. The
PR-gate's pytest step in `.circleci/config.yml` already runs under
`env -i`; this pin enforces the same scrub on the cron path.
"""
from __future__ import annotations
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[4]
RUN_DAILY = REPO_ROOT / "tests" / "e2e" / "claude_code" / "cron_vm" / "run_daily.sh"
def _pytest_invocation_block() -> str:
"""Return only the executable lines around the pytest invocation.
Comment text in run_daily.sh explains *why* certain credential
names must not appear, so a naïve substring scan over the whole
region would false-positive on the rationale itself. Strip lines
whose first non-space character is `#`.
"""
body = RUN_DAILY.read_text()
start = body.index('log "running pytest"')
end = body.index("PYTEST_EXIT=$?", start)
return "\n".join(
line for line in body[start:end].splitlines()
if line.lstrip()[:1] != "#"
)
def test_pytest_invocation_wraps_in_env_i() -> None:
block = _pytest_invocation_block()
assert "env -i" in block, (
"run_daily.sh: the pytest invocation must run under `env -i` so "
"PR-controlled test code under tests/e2e/claude_code/ cannot read "
"provider/agent-shin credentials out of the systemd service "
"environment, and so a model-directed `Read` tool call cannot "
"reach /proc/<pytest-pid>/environ to pull them out."
)
assert block.index("env -i") < block.index('"${WORKTREE_UV}" run pytest'), (
"run_daily.sh: `env -i` must precede the pytest invocation; "
"otherwise pytest inherits the full credential-bearing env."
)
def test_pytest_invocation_env_i_excludes_provider_secrets() -> None:
block = _pytest_invocation_block()
for forbidden in (
"ANTHROPIC_API_KEY",
"AWS_BEARER_TOKEN_BEDROCK",
"AWS_ACCESS_KEY_ID",
"AWS_SECRET_ACCESS_KEY",
"VERTEXAI_CREDENTIALS",
"VERTEXAI_PROJECT",
"VERTEXAI_LOCATION",
"AZURE_FOUNDRY_API_KEY",
"AZURE_FOUNDRY_API_BASE",
"GITHUB_TOKEN",
"AGENT_SHIN_GITHUB_TOKEN",
):
assert forbidden not in block, (
f"run_daily.sh: the pytest-step `env -i` allowlist must not "
f"pass {forbidden} through. Found it inside the pytest "
f"invocation block."
)
def test_pytest_invocation_passes_proxy_url_and_key_explicitly() -> None:
block = _pytest_invocation_block()
assert "LITELLM_PROXY_BASE_URL=" in block, (
"run_daily.sh: the pytest `env -i` block must still pass "
"LITELLM_PROXY_BASE_URL so the test suite knows where to find "
"the loopback proxy."
)
assert "LITELLM_PROXY_API_KEY=" in block, (
"run_daily.sh: the pytest `env -i` block must still pass "
"LITELLM_PROXY_API_KEY so the test suite can authenticate to "
"the loopback proxy."
)
assert "COMPAT_RESULTS_PATH=" in block, (
"run_daily.sh: the pytest `env -i` block must still pass "
"COMPAT_RESULTS_PATH so the conftest writes the per-cell "
"tagged-union artifact to the script-managed path."
)

View file

@ -0,0 +1,289 @@
"""Regression tests for the GitHub release pagination in `run_daily.sh`.
The cron job resolves "newest LiteLLM v*-stable" via the GitHub Releases
API. A previous version of the loop broke as soon as the current page
contained ANY v*-stable tag. The Releases endpoint orders by
`created_at`, NOT by semver, so a backport on an older series cut today
(e.g. v1.80.1-stable) can land on an earlier page than a higher-version
release cut two weeks ago (e.g. v1.83.0-stable). The early-break would
silently pin the cron to a stale tag because the higher-version release
on a later page never made it into the merged set the final `sort_by`
consumed.
These tests pin two things:
1. The buggy early-break-on-first-stable pattern must not return.
2. The loop still terminates early on the standard "empty page" guard
so a quiet release feed doesn't burn API quota.
The shell loop itself is exercised end-to-end with a fake `curl` that
serves canned page JSON, demonstrating that the resolved tag is the
highest-semver stable across all pages even when the highest tag lives
on page 2+.
"""
from __future__ import annotations
import os
import shutil
import subprocess
import textwrap
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[4]
RUN_DAILY = REPO_ROOT / "tests" / "e2e" / "claude_code" / "cron_vm" / "run_daily.sh"
# The extracted snippet starts AFTER `log`/`die` are defined in run_daily.sh,
# so the test harness has to provide its own stubs. Without them, a failure
# inside the snippet (e.g. jq returning an empty LITELLM_VERSION) would crash
# with `bash: die: command not found` (exit 127) instead of the intended
# diagnostic, making test failures unnecessarily hard to debug.
_PREAMBLE = (
"set -Eeuo pipefail\n"
"log() { printf '==> %s\\n' \"$*\" >&2; }\n"
"die() { printf 'ERROR: %s\\n' \"$*\" >&2; exit 1; }\n"
)
def test_run_daily_does_not_early_break_on_first_stable_page() -> None:
"""The regex pattern `select(test("...stable$"))] | length > 0` followed
by `break` is exactly the buggy early-stop. If it ever returns the
cron will silently start testing against a stale stable tag.
"""
body = RUN_DAILY.read_text()
assert (
"length > 0" not in body
or "break" not in body
or (
# If both substrings exist, make sure they aren't both inside the
# same release-pagination loop. The current loop only contains
# a `break` for the empty-page guard, not for any "length > 0"
# condition.
not _shares_loop_body(body, "length > 0", "break")
)
), (
"run_daily.sh contains the old early-break-on-stable pattern. The "
"Releases endpoint orders by created_at, not semver, so breaking "
"on first-stable-seen can miss higher-versioned releases sitting "
"on later pages."
)
def _shares_loop_body(body: str, needle_a: str, needle_b: str) -> bool:
"""Heuristic: do both needles live inside a `for page in ...; do ... done`
block? Used as a defensive guard for the static check above."""
in_loop = False
saw_a = False
saw_b = False
for line in body.splitlines():
stripped = line.strip()
if stripped.startswith("for page in"):
in_loop = True
saw_a = False
saw_b = False
continue
if in_loop and stripped == "done":
if saw_a and saw_b:
return True
in_loop = False
continue
if in_loop:
if needle_a in line:
saw_a = True
if needle_b in line:
saw_b = True
return False
def test_run_daily_keeps_empty_page_break_guard() -> None:
"""The empty-page break is the only break that should remain in the
pagination loop without it a quiet release feed wastes API quota
walking past the last real page."""
body = RUN_DAILY.read_text()
assert "jq 'length' \"${PAGE_JSON}\"" in body, (
"run_daily.sh must still detect empty pages via `jq 'length' "
"${PAGE_JSON}`; without this the loop walks the full 5-page cap "
"even when there are no more releases."
)
assert (
'== "0"' in body
), 'The empty-page guard must compare jq\'s length output to "0".'
def _make_fake_curl(scratch: Path, pages: dict[int, str]) -> Path:
"""Build a fake `curl` shim that serves the canned page JSON for
each `page=N` request and an empty array for any page past the
last canned one.
The shim mimics just enough of curl's CLI surface for the cron
script: it accepts the headers + URL we pass, ignores everything
we don't need, and writes the canned body to either stdout or the
--output target if one is given.
"""
pages_dir = scratch / "pages"
pages_dir.mkdir()
for page_num, body in pages.items():
(pages_dir / f"page{page_num}.json").write_text(body)
curl_path = scratch / "curl"
curl_path.write_text(
textwrap.dedent(
f"""\
#!/usr/bin/env bash
# Fake curl for run_daily.sh release pagination tests. Serves
# page JSON from {pages_dir} keyed by the `page=` query value,
# and returns "[]" for pages past the last canned one (which
# is exactly how the real GitHub API behaves past the end).
url=""
output=""
while [[ $# -gt 0 ]]; do
case "$1" in
-fsS|-fsSL|-H|-o|--output)
if [[ "$1" == "-o" || "$1" == "--output" ]]; then
output="$2"; shift 2
elif [[ "$1" == "-H" ]]; then
shift 2
else
shift
fi
;;
http*)
url="$1"; shift
;;
*)
shift
;;
esac
done
page="$(printf '%s' "$url" | sed -n 's/.*[?&]page=\\([0-9]*\\).*/\\1/p')"
[[ -z "$page" ]] && page=1
file="{pages_dir}/page${{page}}.json"
if [[ -f "$file" ]]; then
if [[ -n "$output" ]]; then cp "$file" "$output"; else cat "$file"; fi
else
if [[ -n "$output" ]]; then printf '[]' > "$output"; else printf '[]'; fi
fi
"""
)
)
curl_path.chmod(0o755)
return curl_path
def _extract_resolution_snippet() -> str:
"""Pull the pagination + sort_by + assignment block out of run_daily.sh
so the test exercises the actual production code path (not a copy).
The block is everything from the GH_AUTH_HEADER setup down through
the LITELLM_VERSION emission.
"""
body = RUN_DAILY.read_text()
start = body.index("GH_AUTH_HEADER=()")
end = body.index('log "resolved litellm:')
return body[start:end]
@pytest.mark.skipif(shutil.which("jq") is None, reason="jq not available")
def test_run_daily_resolves_highest_semver_across_pages(tmp_path: Path) -> None:
"""End-to-end: drive the actual run_daily.sh pagination loop with a
fake curl whose page 1 contains a freshly-cut LOW-version backport
(v1.80.1-stable) and page 2 contains a two-weeks-old HIGH-version
release (v1.83.0-stable). The correct behavior is to resolve
v1.83.0-stable. The pre-fix behavior would resolve v1.80.1-stable
because the early-break consumed only page 1.
"""
pages = {
# Page 1: most-recently-created releases. The order here matches
# what /releases?page=1 returns: created-at descending. The
# freshly-cut v1.80.1-stable backport sits at the top, plus a
# bunch of non-stable releases.
1: """[
{"tag_name": "v1.84.0-nightly.1"},
{"tag_name": "v1.80.1-stable"},
{"tag_name": "v1.84.0-nightly.0"}
]""",
# Page 2: older releases. The HIGHER-version stable lives here
# because it was cut two weeks ago, before the v1.80.1 backport.
2: """[
{"tag_name": "v1.83.0-rc.5"},
{"tag_name": "v1.83.0-stable"},
{"tag_name": "v1.82.4-stable"}
]""",
# Page 3+: empty -> the loop's empty-page guard fires here.
}
fake_curl_dir = tmp_path / "shim"
fake_curl_dir.mkdir()
_make_fake_curl(fake_curl_dir, pages)
workdir = tmp_path / "work"
workdir.mkdir()
snippet = _extract_resolution_snippet()
script = (
_PREAMBLE
+ f"WORKDIR={workdir!s}\n"
+ snippet
+ 'printf "%s" "${LITELLM_VERSION}"\n'
)
env = {
**os.environ,
"PATH": f"{fake_curl_dir}:{os.environ.get('PATH', '')}",
}
# Make sure the loop hits the fake curl, not the system one.
env.pop("GITHUB_TOKEN", None)
result = subprocess.run(
["bash", "-c", script],
capture_output=True,
text=True,
env=env,
check=True,
)
assert result.stdout == "v1.83.0-stable", (
f"Expected the highest-semver stable across pages 1-2, got "
f"{result.stdout!r}. stderr={result.stderr!r}"
)
@pytest.mark.skipif(shutil.which("jq") is None, reason="jq not available")
def test_run_daily_terminates_on_empty_page(tmp_path: Path) -> None:
"""The empty-page guard must fire so we don't always walk all 5
pages. With a single populated page and an empty page 2 we should
stop after fetching page 2 (the first empty response)."""
pages = {1: '[{"tag_name": "v1.50.0-stable"}]'}
fake_curl_dir = tmp_path / "shim"
fake_curl_dir.mkdir()
_make_fake_curl(fake_curl_dir, pages)
workdir = tmp_path / "work"
workdir.mkdir()
snippet = _extract_resolution_snippet()
script = (
_PREAMBLE
+ f"WORKDIR={workdir!s}\n"
+ snippet
+ 'printf "%s" "${LITELLM_VERSION}"\n'
)
env = {
**os.environ,
"PATH": f"{tmp_path}/shim:{os.environ.get('PATH', '')}",
}
env.pop("GITHUB_TOKEN", None)
result = subprocess.run(
["bash", "-c", script],
capture_output=True,
text=True,
env=env,
check=True,
)
assert result.stdout == "v1.50.0-stable"
# Only pages 1 and 2 should have been fetched (2 is empty -> break).
assert (workdir / "releases.page2.json").exists()
assert not (workdir / "releases.page3.json").exists(), (
"Empty-page guard didn't fire — the loop kept walking past the "
"first empty response."
)

View file

@ -0,0 +1,92 @@
"""Pin: the cron `claude --version` probe must run under `env -i`.
The systemd service `litellm-compat-matrix.service` loads provider
credentials (`ANTHROPIC_API_KEY`, `AWS_BEARER_TOKEN_BEDROCK`,
`AZURE_FOUNDRY_API_KEY`) and the agent-shin GitHub token
(`AGENT_SHIN_GITHUB_TOKEN`) into `run_daily.sh`'s environment from
`/etc/litellm-compat-matrix.env`. Running the npm-installed `claude`
binary directly there would hand that full env to package code, so a
compromised `@anthropic-ai/claude-code` release could read those
secrets out of `os.environ` before the proxy or test harness ever
starts. The version probe must be wrapped in `env -i` with a minimal
PATH/HOME/USER/TERM/LANG/LC_ALL/TMPDIR allowlist matching the
PR-gate's resolver/npm-install/pytest scrubs.
"""
from __future__ import annotations
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[4]
RUN_DAILY = REPO_ROOT / "tests" / "e2e" / "claude_code" / "cron_vm" / "run_daily.sh"
def _version_probe_block() -> str:
body = RUN_DAILY.read_text()
start = body.index("CLAUDE_CODE_VERSION=")
end = body.index('[[ -n "${CLAUDE_CODE_VERSION}" ]]', start)
return body[start:end]
def test_version_probe_wraps_claude_in_env_i() -> None:
block = _version_probe_block()
assert "env -i" in block, (
"run_daily.sh: the `claude --version` probe must run under "
"`env -i` so a compromised @anthropic-ai/claude-code package "
"cannot read provider/GitHub credentials out of the systemd "
"service environment."
)
assert block.index("env -i") < block.index("claude --version"), (
"run_daily.sh: `env -i` must precede `claude --version`; "
"otherwise the binary inherits the full credential-bearing env."
)
def test_version_probe_env_i_excludes_provider_secrets() -> None:
block = _version_probe_block()
for forbidden in (
"ANTHROPIC_API_KEY",
"AWS_BEARER_TOKEN_BEDROCK",
"AWS_ACCESS_KEY_ID",
"AWS_SECRET_ACCESS_KEY",
"VERTEXAI_CREDENTIALS",
"AZURE_FOUNDRY_API_KEY",
"GITHUB_TOKEN",
"AGENT_SHIN_GITHUB_TOKEN",
):
assert forbidden not in block, (
f"run_daily.sh: the version-probe `env -i` allowlist must "
f"not pass {forbidden} through. Found it inside the probe "
f"block."
)
def test_version_probe_uses_isolated_home_not_runtime_user_home() -> None:
"""Pin: the `claude --version` probe runs under a fresh empty HOME.
`ProtectHome=read-only` in the systemd unit allows reads of the
runtime user's real home directory. If the probe's `env -i`
block forwards `HOME=${HOME}`, a compromised `claude` package
can `os.path.expanduser("~/.config/gh/hosts.yml")` or
`os.path.expanduser("~/.ssh/...")` and exfiltrate the contents
before the proxy or test harness ever starts. The probe must
point HOME at a per-run tmpdir under `${WORKDIR}` so the CLI
sees an empty HOME instead.
"""
block = _version_probe_block()
body = RUN_DAILY.read_text()
assert "CLAUDE_PROBE_HOME=" in body, (
"run_daily.sh: must define a `CLAUDE_PROBE_HOME` per-run tmpdir "
"for the `claude --version` probe so the CLI never sees the "
"runtime user's real $HOME."
)
assert 'HOME="${CLAUDE_PROBE_HOME}"' in block, (
"run_daily.sh: the probe's `env -i` block must set HOME to "
"the per-run isolated tmpdir, not to the runtime user's $HOME."
)
assert 'HOME="${HOME}"' not in block, (
"run_daily.sh: the probe's `env -i` block must not forward the "
"runtime user's $HOME to `claude --version`. Use the isolated "
"$CLAUDE_PROBE_HOME tmpdir instead."
)

View file

@ -0,0 +1,104 @@
"""Pin: the cron systemd unit hides credential-bearing dotdirs.
`ProtectHome=read-only` blocks writes to /home/mateo but still allows
reads. A model-directed `Read` tool call (the PDF cells pass
`--allowed-tools Read` to the `claude` CLI) or a compromised
`@anthropic-ai/claude-code` package can read absolute paths under
the runtime user's home and exfiltrate the contents — even with the
per-`claude`-invocation HOME isolation in place, because absolute
paths bypass `~`-expansion.
This file pins the second line of defense: the systemd unit lists
the credential-bearing dotdirs (`~/.config/gh`, `~/.ssh`, `~/.aws`,
`~/.docker`, `~/.kube`, `~/.gnupg`) under `InaccessiblePaths=` so
the kernel hides them from every process in the unit's mount
namespace, including any child of `claude --version` or the pytest
run. It also pins that `~/.config/gh` is *not* in `ReadWritePaths=`
we pass `GH_TOKEN` inline to every `gh` invocation in
`run_daily.sh`, so the host gh-cli config is unused.
"""
from __future__ import annotations
import re
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[4]
SERVICE = (
REPO_ROOT / "tests" / "e2e" / "claude_code" / "cron_vm" / "litellm-compat-matrix.service"
)
def _service_text() -> str:
return SERVICE.read_text()
def _directive(name: str) -> str:
"""Return the value of a single-line systemd directive (or empty)."""
text = _service_text()
match = re.search(rf"^\s*{re.escape(name)}\s*=\s*(.*)$", text, re.MULTILINE)
return match.group(1).strip() if match else ""
def test_inaccessible_paths_hides_credential_dotdirs() -> None:
"""Every credential-bearing dotdir must be under `InaccessiblePaths=`."""
inaccessible = _directive("InaccessiblePaths")
assert inaccessible, (
"litellm-compat-matrix.service: must declare `InaccessiblePaths=` "
"to hide credential dotdirs from the `claude` subprocess and the "
"model-directed Read tool. Without this, an absolute-path read "
"like `Read('/home/mateo/.config/gh/hosts.yml')` exfiltrates "
"the gh-cli token despite the per-invocation HOME isolation."
)
for path in (
"/home/mateo/.config/gh",
"/home/mateo/.ssh",
"/home/mateo/.aws",
"/home/mateo/.docker",
"/home/mateo/.kube",
"/home/mateo/.gnupg",
):
# Tolerated `-` prefix means "ignore if missing on host".
assert path in inaccessible, (
f"litellm-compat-matrix.service: `{path}` must appear in "
f"`InaccessiblePaths=` so the cron `claude` subprocess can "
f"never read it (even via an absolute path that bypasses "
f"the per-invocation HOME override)."
)
def test_gh_config_is_not_writeable() -> None:
"""`~/.config/gh` is not whitelisted under `ReadWritePaths=`.
We pass `GH_TOKEN` inline to every `gh` invocation in
`run_daily.sh` (`gh repo clone`, `gh pr create`, `gh pr edit`).
The host `~/.config/gh/hosts.yml` is therefore never consulted
or written to. Keeping it out of `ReadWritePaths=` is the second
line of defense: a future regression that drops the inline-token
convention will fail loudly (gh writes a new login config and
hits a read-only filesystem) rather than silently re-introduce
the credential exfiltration surface that
`InaccessiblePaths=/home/mateo/.config/gh` is closing.
"""
rw = _directive("ReadWritePaths")
assert ".config/gh" not in rw, (
"litellm-compat-matrix.service: `/home/mateo/.config/gh` must "
"*not* appear in `ReadWritePaths=`. We pass `GH_TOKEN` inline "
"to every `gh` invocation in run_daily.sh, so the host gh-cli "
"config is never consulted or written to. Keeping the path out "
"of ReadWritePaths means a future regression that drops the "
"inline-token convention will fail loudly instead of silently "
"re-opening the credential exfiltration surface that "
"`InaccessiblePaths=` is closing."
)
def test_protect_home_is_read_only_or_stricter() -> None:
"""`ProtectHome=` must be at least `read-only`."""
value = _directive("ProtectHome")
assert value in ("read-only", "tmpfs", "yes", "true"), (
f"litellm-compat-matrix.service: `ProtectHome=` must be `read-only`, "
f"`tmpfs`, or `yes`. Got: {value!r}. Without this, the unit can "
f"write anywhere under /home/mateo, including overwriting "
f"~/.config/gh/hosts.yml."
)

View file

@ -0,0 +1,47 @@
"""basic_messaging_non_streaming × Anthropic.
The thinnest end-to-end path through every layer of the matrix: drive the
real `claude` CLI in headless mode against a running LiteLLM proxy that
routes to Anthropic, and report the outcome via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_non_streaming/test_anthropic.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. The shared
`run_basic_messaging_cell` helper fans the three model runs out in
parallel and reports one `compat_result.add(...)` entry per model so
the matrix builder still sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
# Per the PRD: each cell is exercised against three Claude tiers via the
# Anthropic provider. Aliases are configured in the LiteLLM proxy's
# routing config; the driver only sends the alias.
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
def test_basic_messaging_non_streaming_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply.
"Basic messaging" means: send a single user prompt, receive any
non-empty assistant text reply, no tools, no streaming, no thinking.
The whole point of this slice is to prove the path works at all
so the assertion is intentionally lenient on the reply contents.
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=ANTHROPIC_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
)

View file

@ -0,0 +1,53 @@
"""basic_messaging_non_streaming x Azure (Microsoft Foundry).
Drive the real `claude` CLI in headless mode against a running LiteLLM
proxy that routes Claude requests to Anthropic's models hosted in
Microsoft Foundry on Azure, and report the outcome via `compat_result`.
Anthropic announced Claude Haiku 4.5, Sonnet 4.5/4.6, and Opus 4.1/4.6/4.7
in Microsoft Foundry on 2025-11-18; LiteLLM exposes them via the
`azure_ai/claude-*` provider prefix, which talks to Foundry's
Anthropic-shape `/anthropic/v1/messages` endpoint.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_non_streaming/test_azure.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. The shared
`run_basic_messaging_cell` helper fans the three model runs out in
parallel and reports one `compat_result.add(...)` entry per model so
the matrix builder still sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
# Per-model aliases registered in the LiteLLM proxy's routing config to
# point at Microsoft Foundry's Anthropic deployments. The driver only
# sends the alias; the proxy is the one that knows the upstream Foundry
# resource URL and API key.
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
def test_basic_messaging_non_streaming_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply.
"Basic messaging" means: send a single user prompt, receive any
non-empty assistant text reply, no tools, no streaming, no thinking.
The whole point of this slice is to prove the path works at all
so the assertion is intentionally lenient on the reply contents.
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=AZURE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
)

View file

@ -0,0 +1,42 @@
"""basic_messaging_non_streaming x Bedrock (Converse).
Drive the real `claude` CLI in headless mode against a running LiteLLM
proxy that routes Claude requests to AWS Bedrock via the unified
`Converse` API path, and report the outcome via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. The shared
`run_basic_messaging_cell` helper fans the three model runs out in
parallel and reports one `compat_result.add(...)` entry per model so
the matrix builder still sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
# Per-model aliases registered in the LiteLLM proxy's routing config to
# point at Bedrock's Converse endpoint. The driver only sends the alias;
# the proxy is the one that knows the upstream model id and routing
# strategy.
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
def test_basic_messaging_non_streaming_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply."""
run_basic_messaging_cell(
compat_result=compat_result,
models=BEDROCK_CONVERSE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
)

View file

@ -0,0 +1,42 @@
"""basic_messaging_non_streaming x Bedrock (Invoke).
Drive the real `claude` CLI in headless mode against a running LiteLLM
proxy that routes Claude requests to AWS Bedrock via the legacy
`InvokeModel` API path, and report the outcome via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_non_streaming/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. The shared
`run_basic_messaging_cell` helper fans the three model runs out in
parallel and reports one `compat_result.add(...)` entry per model so
the matrix builder still sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
# Per-model aliases registered in the LiteLLM proxy's routing config to
# point at Bedrock's legacy InvokeModel endpoint. The driver only sends
# the alias; the proxy is the one that knows the upstream model id and
# routing strategy.
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
def test_basic_messaging_non_streaming_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply."""
run_basic_messaging_cell(
compat_result=compat_result,
models=BEDROCK_INVOKE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
)

View file

@ -0,0 +1,42 @@
"""basic_messaging_non_streaming x Vertex AI.
Drive the real `claude` CLI in headless mode against a running LiteLLM
proxy that routes Claude requests to Anthropic's models on Google Cloud
Vertex AI, and report the outcome via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_non_streaming/test_vertex_ai.py
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Per the PRD, every cell exercises Claude Haiku 4.5, Sonnet 4.6, and Opus
4.7; the cell only goes green if all three pass. The shared
`run_basic_messaging_cell` helper fans the three model runs out in
parallel and reports one `compat_result.add(...)` entry per model so
the matrix builder still sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
# Per-model aliases registered in the LiteLLM proxy's routing config to
# point at Vertex AI's Anthropic model endpoints. The driver only sends
# the alias; the proxy is the one that knows the upstream publisher
# model id and the GCP region.
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
def test_basic_messaging_non_streaming_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a reply."""
run_basic_messaging_cell(
compat_result=compat_result,
models=VERTEX_AI_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
)

View file

@ -0,0 +1,46 @@
"""basic_messaging_streaming x Anthropic.
Drive the real `claude` CLI in headless `--output-format stream-json
--include-partial-messages` mode against a running LiteLLM proxy that
routes to Anthropic, and report the outcome via `compat_result`.
The cell goes green only when every Claude tier (a) returns a non-empty
reply and (b) the proxy actually streamed it i.e. the CLI observed
multiple `stream_event` records carrying raw SSE deltas. A proxy that
buffers the upstream stream and returns a single non-streaming chunk
emits zero such records, which is the regression this row catches.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_streaming/test_anthropic.py
^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
The shared `run_basic_messaging_cell` helper fans the three Claude tiers
out in parallel inside this single test, with one
`compat_result.add(...)` entry per model so the matrix builder still
sees three rows for this (feature, provider).
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
def test_basic_messaging_streaming_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
non-empty streamed reply (one row per Claude tier).
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=ANTHROPIC_MODELS,
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)

View file

@ -0,0 +1,40 @@
"""basic_messaging_streaming x Azure (Microsoft Foundry).
Drive the real `claude` CLI in headless `--output-format stream-json`
mode against a running LiteLLM proxy that routes Claude requests to
Anthropic's models hosted in Microsoft Foundry on Azure, and report the
outcome via `compat_result`.
Foundry exposes Claude on an Anthropic-shape `/anthropic/v1/messages`
endpoint with native SSE streaming; LiteLLM forwards stream events
through the `azure_ai/claude-*` provider unchanged.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_streaming/test_azure.py
^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
def test_basic_messaging_streaming_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
non-empty streamed reply (one row per Claude tier).
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=AZURE_MODELS,
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)

View file

@ -0,0 +1,36 @@
"""basic_messaging_streaming x Bedrock (Converse).
Drive the real `claude` CLI in headless `--output-format stream-json`
mode against a running LiteLLM proxy that routes Claude requests to AWS
Bedrock via the unified `Converse` API path, and report the outcome via
`compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
def test_basic_messaging_streaming_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
non-empty streamed reply (one row per Claude tier).
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=BEDROCK_CONVERSE_MODELS,
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)

View file

@ -0,0 +1,36 @@
"""basic_messaging_streaming x Bedrock (Invoke).
Drive the real `claude` CLI in headless `--output-format stream-json`
mode against a running LiteLLM proxy that routes Claude requests to AWS
Bedrock via the legacy `InvokeModel` API path, and report the outcome
via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_streaming/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
def test_basic_messaging_streaming_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
non-empty streamed reply (one row per Claude tier).
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=BEDROCK_INVOKE_MODELS,
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)

View file

@ -0,0 +1,36 @@
"""basic_messaging_streaming x Vertex AI.
Drive the real `claude` CLI in headless `--output-format stream-json`
mode against a running LiteLLM proxy that routes Claude requests to
Anthropic's models on Google Cloud Vertex AI, and report the outcome
via `compat_result`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/basic_messaging_streaming/test_vertex_ai.py
^^^^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
from claude_code._basic_messaging import run_basic_messaging_cell
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
def test_basic_messaging_streaming_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
non-empty streamed reply (one row per Claude tier).
"""
run_basic_messaging_cell(
compat_result=compat_result,
models=VERTEX_AI_MODELS,
prompt="Count from 1 to 5, one number per line.",
verify_streaming=True,
)

View file

@ -0,0 +1,542 @@
"""Claude Code CLI Driver.
A thin wrapper around the `claude` CLI in headless mode. Every compatibility
test consumes only this module tests must never shell out directly. This
keeps the subprocess assembly, stream-JSON parsing, and result shape in a
single place that can be unit-tested with a mocked subprocess.
The driver is deliberately small: it knows how to invoke the CLI, drain its
stream-JSON output, and return a structured `DriverResult`. Higher-level
matrix concerns (status aggregation, manifest lookup, JSON serialization)
live in `matrix_builder.py`.
"""
from __future__ import annotations
import json
import os
import shutil
import subprocess
import sys
import tempfile
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import dataclass, field
from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple, Union
from claude_code.rate_limiter import (
RateLimiter,
get_default_limiter,
infer_provider,
)
CLAUDE_CLI_DEFAULT = "claude"
# 120s is fine for a single isolated CLI call against an unloaded
# upstream, but the matrix run launches up to 75 concurrent calls and
# upstreams can take several minutes to respond under that contention.
# We expose the timeout as an env var so binary-search runs can grow
# it without touching the test code.
DEFAULT_TIMEOUT_SECONDS = float(
os.environ.get("LITELLM_COMPAT_CLI_TIMEOUT_SECONDS") or 120
)
# Env vars the `claude` Node CLI legitimately needs to function:
# locating its own binary + node, basic locale/terminal plumbing.
# Deliberately excludes every credential-bearing var that the
# surrounding CI job sets for the proxy (ANTHROPIC_API_KEY, AWS_*,
# AZURE_*, VERTEXAI_CREDENTIALS, GITHUB_TOKEN, OPENAI_API_KEY,
# DATABASE_URL, ...). The CLI talks to the proxy via the explicit
# ANTHROPIC_BASE_URL/ANTHROPIC_AUTH_TOKEN we set below — it has no
# business reading the proxy's upstream credentials, and a
# compromised CLI release shouldn't be able to exfiltrate them out
# of the CI environment.
#
# `HOME` is intentionally NOT in this list: see `_make_isolated_home`
# below. We give the CLI a fresh empty per-invocation HOME so a
# compromised claude package or a model-directed `Read` tool call
# can't reach files like `~/.config/gh/hosts.yml`, `~/.ssh/id_rsa`,
# or `~/.bash_history` on the cron VM (and on the CircleCI executor
# the same isolation prevents accidentally exposing checkout-adjacent
# files even though the runner home is ephemeral there).
_CLI_ENV_ALLOWLIST: tuple = (
"PATH",
"USER",
"LOGNAME",
"SHELL",
"TERM",
"TMPDIR",
"LANG",
"LC_ALL",
"LC_CTYPE",
"NODE_PATH",
"NVM_DIR",
"NVM_BIN",
)
def _make_isolated_home() -> str:
"""Create a fresh empty HOME directory for a single `claude` subprocess.
The CLI needs *a* writable HOME (it caches per-session state under
`$HOME/.claude/projects/<sha>/`), but it has no legitimate need
for the *user's* HOME. Handing it the real one means a compromised
`@anthropic-ai/claude-code` release, or a model-directed `Read`
tool call during a PDF/vision cell, can read host files like
`~/.config/gh/hosts.yml` (GitHub CLI host token), `~/.ssh/`,
`~/.bash_history`, or any other dotfile under the runtime user's
home. On the cron VM the runtime user is a real interactive
account (`mateo`) with a populated home directory, so this is a
real exfiltration surface.
The directory is created under `tempfile.gettempdir()` (which is
`/tmp` on Linux; under systemd's `PrivateTmp=true` that's a
per-service tmpfs that the service user can't otherwise reach).
Caller is responsible for `shutil.rmtree`-ing it after the
subprocess exits.
"""
return tempfile.mkdtemp(prefix="claude-cli-home-")
class ClaudeCLIError(RuntimeError):
"""Raised when the `claude` CLI cannot be invoked or returns a fatal error."""
@dataclass
class DriverResult:
"""Structured outcome of a single `claude` CLI invocation.
`text` is the assistant's final user-visible reply (joined across any
intermediate `assistant` events for non-streaming runs). `events` is the
raw list of stream-JSON objects emitted by the CLI, preserved so test
authors can write feature-specific assertions (tool calls, cache hits,
usage) without re-parsing stdout.
"""
text: str
events: List[Dict[str, Any]] = field(default_factory=list)
exit_code: int = 0
stderr: str = ""
usage: Optional[Dict[str, Any]] = None
duration_ms: Optional[int] = None
def run_claude(
*,
prompt: Optional[str],
model: str,
base_url: str,
api_key: str,
extra_env: Optional[Mapping[str, str]] = None,
extra_args: Optional[Sequence[str]] = None,
stdin_input: Optional[str] = None,
cli_path: str = CLAUDE_CLI_DEFAULT,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
runner: Optional[Any] = None,
rate_limiter: Optional[RateLimiter] = None,
) -> DriverResult:
"""Invoke `claude` once in headless stream-JSON mode and return the result.
The CLI is pointed at a LiteLLM proxy via `ANTHROPIC_BASE_URL` /
`ANTHROPIC_AUTH_TOKEN`, so the same code path exercises every provider
column only the model id and the proxy's routing differ between
invocations.
`runner` is an injection seam used by the unit tests: by default we call
`subprocess.run`, but the test suite swaps in a fake that yields canned
stream-JSON. Production callers should never set it.
`rate_limiter` is the second injection seam: the cross-process
token-bucket limiter throttles outbound calls per provider so a
fully-parallel matrix run doesn't trip 429s. Defaults to the
process-wide singleton; unit tests pass a no-op limiter or one
backed by a tmp dir to keep tests hermetic.
"""
if prompt is None and stdin_input is None:
raise ValueError("must supply either `prompt` or `stdin_input`")
if prompt is not None and stdin_input is not None:
raise ValueError("must supply only one of `prompt` or `stdin_input`, not both")
if prompt is not None and not prompt:
raise ValueError("prompt must be a non-empty string when provided")
if stdin_input is not None and not stdin_input:
raise ValueError("stdin_input must be a non-empty string when provided")
if not model:
raise ValueError("model must be a non-empty string")
if not base_url:
raise ValueError("base_url must be a non-empty string")
if not api_key:
raise ValueError("api_key must be a non-empty string")
# `claude --print` takes the prompt as the **last positional argument**.
# Flags must come before it, otherwise they're parsed as part of the
# prompt (or silently dropped, depending on the CLI version) and the
# tool_use / vision cells fail with confusing "no tool_use observed"
# errors. Build the flag list first, then append the prompt last.
#
# When `extra_args` contains a *variadic* flag like `--allowed-tools
# WebSearch` (commander.js's `<tools...>`), the parser greedily
# consumes every subsequent token as part of the variadic list — so
# the prompt would be eaten as a tool name. Inserting `--` before
# the prompt terminates option parsing and leaves the prompt as a
# plain positional, which works for variadic and non-variadic flags
# alike.
cmd: List[str] = [
cli_path,
"--print",
"--output-format",
"stream-json",
"--verbose",
"--model",
model,
]
if extra_args:
cmd.extend(extra_args)
if prompt is not None:
cmd.append("--")
cmd.append(prompt)
# Build a minimal env for the CLI subprocess: only the allowlisted
# process-runtime vars from os.environ, plus the explicit proxy
# creds, plus any caller-supplied overrides. See _CLI_ENV_ALLOWLIST
# above for the security rationale.
env: Dict[str, str] = {
key: os.environ[key] for key in _CLI_ENV_ALLOWLIST if key in os.environ
}
env["ANTHROPIC_BASE_URL"] = base_url
env["ANTHROPIC_AUTH_TOKEN"] = api_key
# Hand the CLI a fresh empty HOME so a compromised claude package
# or a model-directed Read tool call can't see the runtime user's
# real dotfiles. Created here, removed in the `finally` below
# regardless of how the subprocess exits.
isolated_home = _make_isolated_home()
env["HOME"] = isolated_home
if extra_env:
env.update(extra_env)
# Throttle by provider *before* launching the CLI. Doing this here
# (rather than per-test) means every code path that lands on
# `run_claude` is rate-limited automatically — including
# `run_claude_models_parallel`, which is the hot path during the
# full matrix run.
limiter = rate_limiter if rate_limiter is not None else get_default_limiter()
provider = infer_provider(model)
limiter.acquire(provider)
run_fn = runner or subprocess.run
try:
try:
completed = run_fn(
cmd,
env=env,
input=stdin_input,
capture_output=True,
text=True,
timeout=timeout,
check=False,
)
except FileNotFoundError as exc:
raise ClaudeCLIError(
f"claude CLI not found at {cli_path!r}; install with `npm i -g @anthropic-ai/claude-code`"
) from exc
except subprocess.TimeoutExpired as exc:
raise ClaudeCLIError(f"claude CLI timed out after {timeout}s") from exc
finally:
# Best-effort cleanup. If the subprocess wrote a `.claude/`
# session dir under the isolated HOME, we remove it here so
# parallel matrix runs don't accumulate per-call tmpdirs.
# `ignore_errors=True` because rmtree races with any
# not-yet-reaped child (SIGTERM'd `claude` on a host-side
# timeout) are benign — the next matrix run starts from a
# fresh tmpdir anyway.
shutil.rmtree(isolated_home, ignore_errors=True)
events = _parse_stream_json(completed.stdout or "")
text = _extract_assistant_text(events)
usage = _extract_usage(events)
return DriverResult(
text=text,
events=events,
exit_code=completed.returncode,
stderr=completed.stderr or "",
usage=usage,
)
ModelResult = Union[DriverResult, ClaudeCLIError]
def run_claude_models_parallel(
*,
models: Sequence[str],
prompt: Optional[str],
base_url: str,
api_key: str,
extra_env: Optional[Mapping[str, str]] = None,
extra_args: Optional[Sequence[str]] = None,
stdin_input: Optional[str] = None,
cli_path: str = CLAUDE_CLI_DEFAULT,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
runner: Optional[Callable[..., Any]] = None,
) -> Dict[str, ModelResult]:
"""Invoke `run_claude` for every `models[i]` concurrently and collect outcomes.
Each `claude` CLI invocation is a long-lived subprocess that spends
almost all of its time waiting on the upstream API; running the
three Claude tiers in parallel cuts the per-cell wall time roughly
threefold without changing what each invocation does.
Threads (rather than asyncio) are the right primitive here because
`subprocess.run` releases the GIL while it waits, and we want to
keep the synchronous CLI driver unchanged so unit tests can keep
injecting a fake `runner`.
Returns a dict keyed by model id. Each value is either the
`DriverResult` produced by `run_claude` or the `ClaudeCLIError`
that aborted that model's run — callers decide how to map either
into a `compat_result` entry. The shared kwargs (prompt, env, args,
timeout, runner) are forwarded verbatim so the per-model wire is
identical to what the sequential path produces.
"""
if not models:
raise ValueError("models must be a non-empty sequence")
def _one(model: str) -> Tuple[str, ModelResult, float]:
# Per-model wall clock: this is what the matrix run actually pays for.
# We record it whether the run succeeded or raised so the breakdown
# log below covers both code paths and surfaces "which model is the
# long pole?" without requiring per-test instrumentation.
started = time.monotonic()
try:
result = run_claude(
prompt=prompt,
model=model,
base_url=base_url,
api_key=api_key,
extra_env=extra_env,
extra_args=extra_args,
stdin_input=stdin_input,
cli_path=cli_path,
timeout=timeout,
runner=runner,
)
elapsed = time.monotonic() - started
# Stamp the duration onto the DriverResult so callers (tests,
# diagnostics) can attribute slow cells without re-timing.
result.duration_ms = int(elapsed * 1000)
return model, result, elapsed
except ClaudeCLIError as exc:
elapsed = time.monotonic() - started
return model, exc, elapsed
except Exception as exc:
# Honor the documented "errors as values" contract for any
# exception type — not just ClaudeCLIError. The rate
# limiter does file I/O (OSError), `infer_provider` can
# raise ValueError on edge-case model strings, and a future
# bug elsewhere in the call stack must not abort the entire
# parallel batch and lose the other models' outcomes.
elapsed = time.monotonic() - started
wrapped = ClaudeCLIError(
f"unexpected error running model {model!r}: "
f"{type(exc).__name__}: {exc}"
)
wrapped.__cause__ = exc
return model, wrapped, elapsed
outcomes: Dict[str, ModelResult] = {}
durations: Dict[str, float] = {}
overall_started = time.monotonic()
with ThreadPoolExecutor(max_workers=len(models)) as pool:
futures = [pool.submit(_one, model) for model in models]
for future in as_completed(futures):
model, outcome, elapsed = future.result()
outcomes[model] = outcome
durations[model] = elapsed
overall_elapsed = time.monotonic() - overall_started
_log_parallel_breakdown(models, durations, outcomes, overall_elapsed)
return outcomes
def _log_parallel_breakdown(
models: Sequence[str],
durations: Mapping[str, float],
outcomes: Mapping[str, ModelResult],
overall_elapsed: float,
) -> None:
"""Emit a one-block timing breakdown to stderr.
Pytest only shows captured output for failing tests by default, but
`-s` surfaces it for passing tests too which is exactly when you
care about "did parallelization actually help?". The block reports:
- per-model wall time and outcome (ok / cli-error / non-zero exit)
- the slowest model (the parallel run's wall-time floor)
- the sum of sequential model times (what the old serial path
would have paid)
- the overall parallel wall time and the speedup ratio
If one model dominates, `slowest overall sequential / 1`, and
the speedup will be near 1× exactly the diagnostic that explains
"why didn't this get faster?".
"""
sequential_total = sum(durations.values())
slowest_model = max(durations, key=durations.get) if durations else None
slowest = durations[slowest_model] if slowest_model else 0.0
speedup = sequential_total / overall_elapsed if overall_elapsed > 0 else 0.0
lines: List[str] = []
lines.append("[parallel] per-model wall time:")
for model in models:
elapsed = durations.get(model, 0.0)
outcome = outcomes.get(model)
if isinstance(outcome, ClaudeCLIError):
status = "cli-error"
elif isinstance(outcome, DriverResult):
status = f"exit={outcome.exit_code}"
else:
status = "missing"
lines.append(f" {model:<40s} {elapsed:6.2f}s ({status})")
if slowest_model is not None:
lines.append(
f"[parallel] slowest={slowest_model} ({slowest:.2f}s); "
f"sequential_sum={sequential_total:.2f}s; "
f"parallel_wall={overall_elapsed:.2f}s; "
f"speedup={speedup:.2f}x"
)
print("\n".join(lines), file=sys.stderr, flush=True)
def _parse_stream_json(stdout: str) -> List[Dict[str, Any]]:
"""Parse newline-delimited JSON emitted by `claude --output-format stream-json`.
Lines that don't parse as JSON are silently skipped — the CLI occasionally
emits debug output we don't care about, and a single malformed line should
not abort the whole run. Real failure modes surface via exit code.
"""
events: List[Dict[str, Any]] = []
for line in stdout.splitlines():
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(obj, dict):
events.append(obj)
return events
def _extract_assistant_text(events: Sequence[Mapping[str, Any]]) -> str:
"""Concatenate the text content of every `assistant` event in order.
The non-streaming `--print` path emits a single `assistant` event whose
`message.content` is a list of content blocks. We walk the blocks and
join every `text` block the CLI prints other block types (e.g.
`tool_use`) which we ignore for the basic-messaging case.
"""
chunks: List[str] = []
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if isinstance(content, str):
chunks.append(content)
continue
if not isinstance(content, list):
continue
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") == "text" and isinstance(block.get("text"), str):
chunks.append(block["text"])
return "".join(chunks)
def failure_diagnostic(result: "DriverResult", *, max_len: int = 800) -> str:
"""Build a human-readable error string from a non-zero `claude` CLI run.
The CLI is annoying to debug because the most useful failure signal
rarely lands on stderr. When the proxy returns an HTTP error, the CLI
swallows it into an `assistant`/`result` event on **stdout** with
`is_error: true` and a JSON-shaped `text` block and exits non-zero.
Tests that only print `stderr.strip()` see an empty string, which is
exactly the situation that masked a misconfigured proxy in early
bring-up of the compat matrix.
This helper concatenates the most useful diagnostic we can find, in
priority order:
1. `result.text` (the assistant's user-visible reply, which is where
API errors land in stream-json mode), trimmed
2. `api_error_status` from any `result` event, if present
3. `result.stderr`, trimmed
4. `<no diagnostic output>` as a last resort
The output is truncated to `max_len` characters so a giant HTML 502
page from a misbehaving load balancer doesn't blow up the matrix
JSON.
"""
pieces: List[str] = [f"exit={result.exit_code}"]
# api_error_status only appears on the final `result` event when the
# CLI received an HTTP error from the upstream API. Surfacing it
# explicitly makes "is this a proxy/auth problem or a CLI problem?"
# answerable without re-reading the events list.
api_status = _extract_api_error_status(result.events)
if api_status is not None:
pieces.append(f"api_status={api_status}")
text = (result.text or "").strip()
if text:
pieces.append(f"text={_truncate(text, max_len)}")
stderr = (result.stderr or "").strip()
if stderr:
pieces.append(f"stderr={_truncate(stderr, max_len)}")
if len(pieces) == 1:
pieces.append("(no diagnostic output)")
return "; ".join(pieces)
def _extract_api_error_status(
events: Sequence[Mapping[str, Any]],
) -> Optional[int]:
"""Return the `api_error_status` from the last `result` event, if any."""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
status = event.get("api_error_status")
if isinstance(status, int):
return status
return None
def _truncate(s: str, max_len: int) -> str:
if len(s) <= max_len:
return s
return s[:max_len] + "...(truncated)"
def _extract_usage(events: Sequence[Mapping[str, Any]]) -> Optional[Dict[str, Any]]:
"""Return the most recent `usage` block seen on any event, if any.
The CLI surfaces token + cache usage on the final `result` event for
non-streaming runs, but earlier events also carry partial usage in some
versions; taking the last non-empty one is the safe default.
"""
last: Optional[Dict[str, Any]] = None
for event in events:
usage = event.get("usage")
if isinstance(usage, dict) and usage:
last = usage
continue
message = event.get("message")
if isinstance(message, dict):
inner = message.get("usage")
if isinstance(inner, dict) and inner:
last = inner
return last

View file

@ -0,0 +1,550 @@
"""Pytest plumbing for the Claude Code compatibility matrix.
Three responsibilities live here:
1. The `compat_result` fixture the only API a test author needs to learn.
Tests call `compat_result.set({"status": "pass"})` (or fail / not_applicable)
to report their outcome as a tagged union. Multi-model tests call
`.add(...)` once per Claude tier so each tier lands as its own row in
the results artifact.
2. The `pytest_runtest_makereport` hook captures each test's reported result,
infers (feature, provider) from the file path, and accumulates rows into
a per-process collector. At session end we serialize them to
`compat-results.json` (or a per-worker file under xdist) so the Matrix
JSON Builder can consume them.
3. xdist coordination when `pytest -n auto` is used, every worker writes
its own results shard and the controller merges them into the canonical
`compat-results.json` in `pytest_sessionfinish`. Without this, the
workers race on the same path and the artifact only reflects whichever
worker finished last. The same merge step also emits a rate-limit
summary that the binary-search helper consumes to decide whether the
current X/Y/Z values were too aggressive.
The (feature, provider) inference comes from the test file path: the parent
directory name is the feature_id (matching `manifest.yaml`), and the file
stem after the leading `test_` is the provider id. This avoids per-file
metadata that drifts.
"""
from __future__ import annotations
import functools
import json
import os
import re
import sys
from collections import Counter, defaultdict
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, FrozenSet, List, Optional, Tuple
import pytest
import yaml
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
RESULTS_ARTIFACT_ENV = "COMPAT_RESULTS_PATH"
DEFAULT_ARTIFACT_PATH = "compat-results.json"
RATE_LIMIT_SUMMARY_ENV = "COMPAT_RATE_LIMIT_SUMMARY_PATH"
DEFAULT_RATE_LIMIT_SUMMARY_PATH = "compat-rate-limit-summary.json"
# Heuristic: detect 429s and rate-limit-shaped errors anywhere in the
# error string. The CLI buries upstream errors in `assistant.message.content`
# text on stdout (see `failure_diagnostic`), so we don't get a structured
# status code in every code path — a regex over the joined error text is
# the most reliable signal we have.
#
# We also treat a CLI timeout (`claude CLI timed out after Ns`) as a
# rate-limit-shaped failure for binary-search purposes: in practice the
# only reason every model in a cell stalls past the timeout is the
# upstream collapsing under concurrency, which is exactly the situation
# the rate limiter is supposed to back off from. False positives on a
# genuinely slow upstream are tolerable here because the worst case is
# the binary search runs at a slightly lower rate than necessary.
_RATE_LIMIT_RE = re.compile(
r"(?:\b429\b|rate[\s_-]?limit|too\s+many\s+requests|throttl(?:ed|ing)|"
r"claude\s+CLI\s+timed\s+out)",
re.IGNORECASE,
)
@dataclass
class CompatResult:
"""Per-test recorder for compatibility outcomes.
Tests interact via `.set(...)` (single result) or `.add(...)` (one
result per Claude tier when the test fans the three models out in
parallel). `.value` and `.values` are read by the
`pytest_runtest_makereport` hook after the test body finishes.
Multi-result usage exists because every cell in the compat matrix is
backed by three model invocations (Haiku/Sonnet/Opus) per (feature,
provider). When a test runs them concurrently in a single pytest
node, each model needs its own entry in the results artifact so the
matrix builder's per-cell aggregator can apply its "all three must
pass" rule.
"""
value: Optional[Dict[str, Any]] = None
values: List[Dict[str, Any]] = field(default_factory=list)
def set(self, result: Dict[str, Any]) -> None:
validated = self._validate(result)
self.value = validated
def add(self, result: Dict[str, Any]) -> None:
"""Append one model's outcome to the per-test results list.
Use this when a single test exercises multiple Claude tiers
concurrently and needs to report one outcome per tier. The
conftest hook will emit one entry per appended result.
"""
validated = self._validate(result)
self.values.append(validated)
@staticmethod
def _validate(result: Dict[str, Any]) -> Dict[str, Any]:
if not isinstance(result, dict):
raise TypeError("compat_result requires a dict")
status = result.get("status")
if status not in VALID_STATUSES:
raise ValueError(
f"compat_result status must be one of {sorted(VALID_STATUSES)}, "
f"got {status!r}"
)
if status == "fail" and not result.get("error"):
raise ValueError("compat_result {'status': 'fail'} requires 'error'")
if status == "not_applicable" and not result.get("reason"):
raise ValueError(
"compat_result {'status': 'not_applicable'} requires 'reason'"
)
return dict(result)
def collected(self) -> List[Dict[str, Any]]:
"""Return every result reported during the test, preserving order.
Multi-model tests use `.add(...)` per model; legacy tests use
`.set(...)` once. We surface both shapes in a single list so
the makereport hook only has to think about a list of results.
"""
if self.values:
return list(self.values)
if self.value is not None:
return [dict(self.value)]
return []
@dataclass
class _CollectedResult:
feature_id: str
provider: str
nodeid: str
result: Dict[str, Any]
@dataclass
class _Collector:
items: List[_CollectedResult] = field(default_factory=list)
_COLLECTOR = _Collector()
@pytest.fixture
def compat_result() -> CompatResult:
"""Per-test recorder for the (feature, provider) outcome.
Tests should call `compat_result.set({"status": "pass"})` (or fail /
not_applicable) before returning. If a test exits without calling `.set()`
the harness records `status="fail"` with an explanatory error so that
every collected node maps to a real cell.
"""
return CompatResult()
@functools.lru_cache(maxsize=1)
def _manifest_feature_ids() -> FrozenSet[str]:
"""Return the set of feature_ids declared in `manifest.yaml`.
Used as a positive filter so only directories that correspond to a
real matrix row contribute results utility/support directories
(e.g. `cron_vm`, `_driver_unit_tests`) are dropped regardless of
naming convention, and the rate-limit summary stays clean.
Returns an empty set if the manifest is missing or malformed; the
caller treats that as "no path is a feature path", which is the
safe default we'd rather drop a real result than pollute the
artifact with a garbage cell.
"""
manifest_path = Path(__file__).resolve().parent / "manifest.yaml"
try:
raw = yaml.safe_load(manifest_path.read_text())
except (OSError, yaml.YAMLError):
return frozenset()
if not isinstance(raw, dict):
return frozenset()
features = raw.get("features")
if not isinstance(features, list):
return frozenset()
return frozenset(
entry["id"]
for entry in features
if isinstance(entry, dict) and isinstance(entry.get("id"), str)
)
def _infer_feature_and_provider(node_path: Path) -> Optional[tuple]:
"""Infer (feature_id, provider) from a test file path.
Path shape: tests/e2e/claude_code/<feature_id>/test_<provider>.py
Returns None if the file is not a per-feature test (e.g. unit tests
under `_driver_unit_tests/` or support code under `cron_vm/`), so
those don't pollute the matrix artifact. We positively filter the
parent directory against `manifest.yaml` rather than relying on
naming conventions, because non-feature siblings don't all share
an underscore prefix.
"""
name = node_path.name
if not name.startswith("test_") or not name.endswith(".py"):
return None
provider = name[len("test_") : -len(".py")]
feature_id = node_path.parent.name
if feature_id not in _manifest_feature_ids():
return None
return feature_id, provider
@pytest.hookimpl(hookwrapper=True)
def pytest_runtest_makereport(item, call):
"""Capture compat_result reports at end-of-test and remember them for the artifact.
A single test may report multiple results (one per Claude tier when
the three are run in parallel inside one node). We emit one
`_CollectedResult` per reported entry so the matrix builder's
per-cell aggregator sees the same shape it would have seen if the
test were parametrized every model lands in the artifact.
"""
outcome = yield
report = outcome.get_result()
# We record on two phases:
# - "call": the normal end-of-test path.
# - "setup" but only on failure: fixture/import errors that prevent the
# test body from running. Without recording these, a broken setup
# silently becomes "not_tested" in the published matrix instead of
# "fail". Teardown is ignored — by then "call" already recorded the
# outcome, and a teardown-only failure (e.g. fixture finalizer) is
# not a cell-level signal.
if report.when == "setup":
if not report.failed:
return
elif report.when != "call":
return
# A skipped test (e.g. `pytest.skip(...)` called inside the body or
# by a `pytest.mark.skipif` evaluated at call time) is neither a
# pass nor a fail — it just didn't run. Recording it as anything
# here would produce a spurious row (the not-failed/empty-collected
# branch below would mark it as a fail with "test passed without
# reporting via compat_result"), so bail out and let the cell stay
# "not_tested" in the published matrix.
if report.skipped:
return
inferred = _infer_feature_and_provider(Path(str(item.path)))
if inferred is None:
return
feature_id, provider = inferred
fixture = item.funcargs.get("compat_result") if hasattr(item, "funcargs") else None
collected: List[Dict[str, Any]] = (
fixture.collected() if isinstance(fixture, CompatResult) else []
)
if report.failed and not any(entry.get("status") == "fail" for entry in collected):
# The test body (or setup) raised and the test author hasn't
# already recorded a fail row via `.add(...)`. If the test had
# recorded only per-model passes before crashing, those partial
# entries would aggregate to "pass" and hide the crash from the
# published matrix; append an explicit "fail" row so the cell
# aggregator (which gives precedence to any fail) surfaces the
# breakage. We skip the append when a fail row is already
# present so that the common pattern — `.add({"status": "fail",
# ...})` per failing model, then `pytest.fail("; ".join(...))`
# to surface them — doesn't produce a phantom duplicate row.
collected = collected + [
{
"status": "fail",
"error": (str(report.longrepr) if report.longrepr else "test failed"),
}
]
elif not report.failed and not collected:
collected = [
{
"status": "fail",
"error": "test passed without reporting via compat_result; "
"every compat test must report a status.",
}
]
for reported in collected:
_COLLECTOR.items.append(
_CollectedResult(
feature_id=feature_id,
provider=provider,
nodeid=report.nodeid,
result=reported,
)
)
def _is_xdist_worker(session) -> bool:
"""Return True iff the current pytest session is an xdist worker.
The standard idiom is to look up `workerinput` on the config; the
controller process doesn't have it, the workers do. We deliberately
don't `import xdist` because the suite must keep running when xdist
isn't installed at all.
"""
return hasattr(session.config, "workerinput")
def _xdist_worker_id(session) -> Optional[str]:
info = getattr(session.config, "workerinput", None)
if not info:
return None
return info.get("workerid")
def _shard_dir(artifact_path: Path) -> Path:
"""Workers write their shards next to the canonical results path.
Putting shards in a sibling directory (rather than inline JSON
files in the same dir) keeps the controller's merge step simple
it just lists `*.json` in `<artifact>.shards/` and avoids
accidental shard/canonical filename collisions.
"""
return artifact_path.with_name(artifact_path.name + ".shards")
def _serialize_items(items: List["_CollectedResult"]) -> List[Dict[str, Any]]:
return [
{
"feature_id": item.feature_id,
"provider": item.provider,
"nodeid": item.nodeid,
"result": item.result,
}
for item in items
]
def _build_rate_limit_summary(
rows: List[Dict[str, Any]],
) -> Dict[str, Any]:
"""Aggregate per-provider rate-limit signals from the result rows.
We classify any failure whose error string matches `_RATE_LIMIT_RE`
as a "rate-limited" failure. The binary-search helper reads this
summary to decide whether the current X/Y/Z values were too
aggressive: if any provider has `rate_limited > 0`, the harness
should back off that provider's rate and retry.
Returns a dict shaped:
{
"totals": {"pass": N, "fail": N, "rate_limited": N, ...},
"per_provider": {
"anthropic": {"pass": ..., "fail": ..., "rate_limited": ...},
...
},
"rate_limited_examples": [
{"feature_id": ..., "provider": ..., "error": "..."}, ...
],
}
"""
totals: Counter = Counter()
per_provider: Dict[str, Counter] = defaultdict(Counter)
rate_limited_examples: List[Dict[str, Any]] = []
for row in rows:
result = row.get("result") or {}
status = result.get("status") or "unknown"
provider = row.get("provider") or "unknown"
totals[status] += 1
per_provider[provider][status] += 1
if status == "fail":
error = str(result.get("error") or "")
if _RATE_LIMIT_RE.search(error):
totals["rate_limited"] += 1
per_provider[provider]["rate_limited"] += 1
# Cap examples so a stuck-throttled run doesn't write
# a multi-MB summary file the helper has to slurp.
if len(rate_limited_examples) < 25:
rate_limited_examples.append(
{
"feature_id": row.get("feature_id"),
"provider": provider,
"error": error[:500],
}
)
return {
"totals": dict(totals),
"per_provider": {p: dict(c) for p, c in per_provider.items()},
"rate_limited_examples": rate_limited_examples,
}
def _print_rate_limit_summary(summary: Dict[str, Any]) -> None:
"""Emit a human-readable per-provider table to stderr.
Pytest only captures stderr when `-s` isn't set; we deliberately
write here anyway because the binary-search workflow runs pytest
with `-q` and grep-checks the structured JSON artifact, while a
human running locally with `-s` sees the same numbers inline.
"""
totals = summary.get("totals", {})
per_provider = summary.get("per_provider", {})
lines: List[str] = []
lines.append("[compat] session totals:")
for status in ("pass", "fail", "rate_limited", "not_applicable", "not_tested"):
if status in totals:
lines.append(f" {status:<16s} {totals[status]}")
if per_provider:
lines.append("[compat] per-provider breakdown:")
for provider in sorted(per_provider):
counts = per_provider[provider]
parts = " ".join(
f"{k}={v}"
for k, v in sorted(counts.items())
if k != "not_tested" or v > 0
)
lines.append(f" {provider:<20s} {parts}")
if totals.get("rate_limited", 0):
lines.append(
"[compat] WARNING: at least one cell hit a rate-limit-shaped error; "
"lower the corresponding LITELLM_COMPAT_RATE_<PROVIDER> and retry"
)
print("\n".join(lines), file=sys.stderr, flush=True)
def pytest_sessionstart(session):
"""Reset per-session state before tests run.
Two responsibilities:
1. Clear the module-level `_COLLECTOR` singleton, which survives
across `pytest.main()` invocations within the same Python
process. Without this reset, results from a prior session
would leak into the next run's `compat-results.json` artifact.
2. Remove stale per-worker shards from any prior session. Without
this, a previous run's shard directory leaks into the next
`pytest_sessionfinish` merge yielding a `compat-results.json`
that includes results from runs that aren't part of the current
session, and a misleading rate-limit summary that re-flags
failures the user already saw and addressed. Only the
controller (non-xdist-worker) clears; workers must not race
the controller while it's wiping the directory.
"""
_COLLECTOR.items.clear()
_manifest_feature_ids.cache_clear()
if _is_xdist_worker(session):
return
artifact_path = Path(os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH)
shard_dir = _shard_dir(artifact_path)
if not shard_dir.exists():
return
for stale in shard_dir.glob("*.json"):
try:
stale.unlink()
except OSError:
# If we can't remove a stale shard (permissions, race with
# an unrelated process), keep going — the merge step is
# robust to malformed shards, and a stale row landing in
# the artifact is recoverable; aborting the session isn't.
continue
def pytest_sessionfinish(session, exitstatus):
"""Write the per-process results shard, then merge if we're the controller.
Worker processes (xdist `gw0`, `gw1`, ...) only write their shard
under `<artifact>.shards/<workerid>.json`. The controller writes
its own shard if it ran any tests itself, then walks the shards
directory and produces the canonical `compat-results.json` plus
the rate-limit summary. Single-process runs (no xdist) take the
same code path with a single shard, so behavior is consistent.
Skip when no compat results were collected this conftest is
loaded for every test under `tests/e2e/claude_code/`, including sibling
unit-test trees (e.g. `_driver_unit_tests/`). Writing an empty
artifact would silently overwrite a real artifact from a prior
compat-test run on the same checkout.
The xdist controller hits this hook with `_COLLECTOR.items` empty
(it never executes tests itself) and `_is_xdist_worker` False, so
we additionally allow the merge step to run when worker shards
are already on disk otherwise the canonical artifact would
never be produced under `pytest -n auto`.
"""
artifact_path = Path(os.environ.get(RESULTS_ARTIFACT_ENV) or DEFAULT_ARTIFACT_PATH)
shard_dir = _shard_dir(artifact_path)
has_worker_shards = shard_dir.is_dir() and any(shard_dir.glob("*.json"))
if not _COLLECTOR.items and not _is_xdist_worker(session) and not has_worker_shards:
return
shard_dir.mkdir(parents=True, exist_ok=True)
worker_id = _xdist_worker_id(session) or "main"
shard_path = shard_dir / f"{worker_id}.json"
shard_path.write_text(
json.dumps(
{
"schema_version": "1",
"worker_id": worker_id,
"results": _serialize_items(_COLLECTOR.items),
},
indent=2,
sort_keys=True,
)
)
# Workers stop here. The controller merges; if we're not running
# under xdist, we are effectively the controller.
if _is_xdist_worker(session):
return
merged_rows: List[Dict[str, Any]] = []
for shard_file in sorted(shard_dir.glob("*.json")):
try:
shard = json.loads(shard_file.read_text())
except (OSError, ValueError):
continue
rows = shard.get("results")
if isinstance(rows, list):
merged_rows.extend(rows)
# Skip writing artifact + summary entirely for unit-test-only runs
# (no per-feature compat rows). Otherwise every `pytest tests/...`
# run — including local unit-test invocations — would silently
# overwrite a real artifact from a prior compat-test run.
if not merged_rows:
return
artifact_path.write_text(
json.dumps(
{"schema_version": "1", "results": merged_rows},
indent=2,
sort_keys=True,
)
)
summary = _build_rate_limit_summary(merged_rows)
summary_path = Path(
os.environ.get(RATE_LIMIT_SUMMARY_ENV) or DEFAULT_RATE_LIMIT_SUMMARY_PATH
)
summary_path.write_text(json.dumps(summary, indent=2, sort_keys=True))
_print_rate_limit_summary(summary)

View file

@ -0,0 +1,94 @@
"""count_tokens x Anthropic.
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
the `claude` CLI: it `POST`s directly to
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
the response is shaped `{"input_tokens": <positive int>}`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/count_tokens/test_anthropic.py
^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Why HTTP probe instead of CLI:
Claude Code calls `count_tokens` internally to compute budget /
context-window usage display, but the result is consumed by the CLI
in-process and never appears in stream-json events. There is no CLI
flag that emits the count to stdout in a way our existing
stream-json parser can pick up, so we can't test the endpoint round
trip through the CLI surface.
The proxy *is* expected to expose `/v1/messages/count_tokens` for
every Claude-style provider it routes to -- LiteLLM has historically
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
Treating it as a matrix row keeps regressions in the cron's daily
diff.
The cell goes red if *any* tier's probe fails the minimal shape
check; the matrix's per-cell aggregator handles that automatically.
Three tiers run sequentially because count_tokens is cheap (<100ms
per request typical) and the parallelization that matters for the
CLI rows isn't useful here.
"""
from __future__ import annotations
import os
import pytest
from claude_code.http_probe import (
assert_count_tokens_shape,
probe_count_tokens,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
def test_count_tokens_anthropic(compat_result):
"""Probe `/v1/messages/count_tokens` for each Anthropic tier and
assert the response shape."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
failures = []
for model in ANTHROPIC_MODELS:
result = probe_count_tokens(
base_url=base_url, api_key=api_key, model=model
)
shape_error = assert_count_tokens_shape(result)
if shape_error is not None:
error = f"[{model}] count_tokens probe failed: {shape_error}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,94 @@
"""count_tokens x Azure (Microsoft Foundry).
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
the `claude` CLI: it `POST`s directly to
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
the response is shaped `{"input_tokens": <positive int>}`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/count_tokens/test_azure.py
^^^^^^^^^^^^ ^^^^^
feature_id provider
Why HTTP probe instead of CLI:
Claude Code calls `count_tokens` internally to compute budget /
context-window usage display, but the result is consumed by the CLI
in-process and never appears in stream-json events. There is no CLI
flag that emits the count to stdout in a way our existing
stream-json parser can pick up, so we can't test the endpoint round
trip through the CLI surface.
The proxy *is* expected to expose `/v1/messages/count_tokens` for
every Claude-style provider it routes to -- LiteLLM has historically
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
Treating it as a matrix row keeps regressions in the cron's daily
diff.
The cell goes red if *any* tier's probe fails the minimal shape
check; the matrix's per-cell aggregator handles that automatically.
Three tiers run sequentially because count_tokens is cheap (<100ms
per request typical) and the parallelization that matters for the
CLI rows isn't useful here.
"""
from __future__ import annotations
import os
import pytest
from claude_code.http_probe import (
assert_count_tokens_shape,
probe_count_tokens,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
def test_count_tokens_azure(compat_result):
"""Probe `/v1/messages/count_tokens` for each Azure (Microsoft Foundry) tier and
assert the response shape."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
failures = []
for model in AZURE_MODELS:
result = probe_count_tokens(
base_url=base_url, api_key=api_key, model=model
)
shape_error = assert_count_tokens_shape(result)
if shape_error is not None:
error = f"[{model}] count_tokens probe failed: {shape_error}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,94 @@
"""count_tokens x Bedrock (Converse).
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
the `claude` CLI: it `POST`s directly to
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
the response is shaped `{"input_tokens": <positive int>}`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/count_tokens/test_bedrock_converse.py
^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
Why HTTP probe instead of CLI:
Claude Code calls `count_tokens` internally to compute budget /
context-window usage display, but the result is consumed by the CLI
in-process and never appears in stream-json events. There is no CLI
flag that emits the count to stdout in a way our existing
stream-json parser can pick up, so we can't test the endpoint round
trip through the CLI surface.
The proxy *is* expected to expose `/v1/messages/count_tokens` for
every Claude-style provider it routes to -- LiteLLM has historically
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
Treating it as a matrix row keeps regressions in the cron's daily
diff.
The cell goes red if *any* tier's probe fails the minimal shape
check; the matrix's per-cell aggregator handles that automatically.
Three tiers run sequentially because count_tokens is cheap (<100ms
per request typical) and the parallelization that matters for the
CLI rows isn't useful here.
"""
from __future__ import annotations
import os
import pytest
from claude_code.http_probe import (
assert_count_tokens_shape,
probe_count_tokens,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
def test_count_tokens_bedrock_converse(compat_result):
"""Probe `/v1/messages/count_tokens` for each Bedrock (Converse) tier and
assert the response shape."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
result = probe_count_tokens(
base_url=base_url, api_key=api_key, model=model
)
shape_error = assert_count_tokens_shape(result)
if shape_error is not None:
error = f"[{model}] count_tokens probe failed: {shape_error}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,94 @@
"""count_tokens x Bedrock (Invoke).
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
the `claude` CLI: it `POST`s directly to
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
the response is shaped `{"input_tokens": <positive int>}`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/count_tokens/test_bedrock_invoke.py
^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
Why HTTP probe instead of CLI:
Claude Code calls `count_tokens` internally to compute budget /
context-window usage display, but the result is consumed by the CLI
in-process and never appears in stream-json events. There is no CLI
flag that emits the count to stdout in a way our existing
stream-json parser can pick up, so we can't test the endpoint round
trip through the CLI surface.
The proxy *is* expected to expose `/v1/messages/count_tokens` for
every Claude-style provider it routes to -- LiteLLM has historically
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
Treating it as a matrix row keeps regressions in the cron's daily
diff.
The cell goes red if *any* tier's probe fails the minimal shape
check; the matrix's per-cell aggregator handles that automatically.
Three tiers run sequentially because count_tokens is cheap (<100ms
per request typical) and the parallelization that matters for the
CLI rows isn't useful here.
"""
from __future__ import annotations
import os
import pytest
from claude_code.http_probe import (
assert_count_tokens_shape,
probe_count_tokens,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
def test_count_tokens_bedrock_invoke(compat_result):
"""Probe `/v1/messages/count_tokens` for each Bedrock (Invoke) tier and
assert the response shape."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
result = probe_count_tokens(
base_url=base_url, api_key=api_key, model=model
)
shape_error = assert_count_tokens_shape(result)
if shape_error is not None:
error = f"[{model}] count_tokens probe failed: {shape_error}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,94 @@
"""count_tokens x Vertex AI.
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
the `claude` CLI: it `POST`s directly to
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
the response is shaped `{"input_tokens": <positive int>}`.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/count_tokens/test_vertex_ai.py
^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Why HTTP probe instead of CLI:
Claude Code calls `count_tokens` internally to compute budget /
context-window usage display, but the result is consumed by the CLI
in-process and never appears in stream-json events. There is no CLI
flag that emits the count to stdout in a way our existing
stream-json parser can pick up, so we can't test the endpoint round
trip through the CLI surface.
The proxy *is* expected to expose `/v1/messages/count_tokens` for
every Claude-style provider it routes to -- LiteLLM has historically
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
Treating it as a matrix row keeps regressions in the cron's daily
diff.
The cell goes red if *any* tier's probe fails the minimal shape
check; the matrix's per-cell aggregator handles that automatically.
Three tiers run sequentially because count_tokens is cheap (<100ms
per request typical) and the parallelization that matters for the
CLI rows isn't useful here.
"""
from __future__ import annotations
import os
import pytest
from claude_code.http_probe import (
assert_count_tokens_shape,
probe_count_tokens,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
def test_count_tokens_vertex_ai(compat_result):
"""Probe `/v1/messages/count_tokens` for each Vertex AI tier and
assert the response shape."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
failures = []
for model in VERTEX_AI_MODELS:
result = probe_count_tokens(
base_url=base_url, api_key=api_key, model=model
)
shape_error = assert_count_tokens_shape(result)
if shape_error is not None:
error = f"[{model}] count_tokens probe failed: {shape_error}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,50 @@
"""Tiny CLI wrapper around `claude_code.matrix_builder.build_from_paths`.
Exists only so `run_daily.sh` can hand the version metadata + paths into
the matrix builder without re-implementing it in bash. All real logic
lives in `matrix_builder.py`, which has its own unit tests under
`_builder_unit_tests/`.
Invoked from the cron worktree (where `uv sync` has installed pyyaml),
not the dev checkout the bash script `cd`s into the worktree before
`uv run python`-ing this file.
"""
from __future__ import annotations
import argparse
import datetime
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
from claude_code.matrix_builder import build_from_paths # noqa: E402 # import needs the sys.path bootstrap above
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--results", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--litellm-version", required=True)
parser.add_argument("--claude-code-version", required=True)
args = parser.parse_args()
generated_at = datetime.datetime.now(datetime.timezone.utc).strftime(
"%Y-%m-%dT%H:%M:%SZ"
)
build_from_paths(
manifest_path=args.manifest,
results_path=args.results,
litellm_version=args.litellm_version,
claude_code_version=args.claude_code_version,
generated_at=generated_at,
output_path=args.output,
)
print(f"wrote {args.output}")
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,50 @@
# Environment file consumed by `litellm-compat-matrix.service`.
#
# Install at `/etc/litellm-compat-matrix.env` and chmod 0600.
# `EnvironmentFile=-` in the unit means the service is allowed to start
# even if this file is missing, but the populator will fail at the
# first provider request without these credentials.
# Anthropic
ANTHROPIC_API_KEY=
# Bedrock (invoke + converse columns).
# Use Anthropic's Bedrock API-key passthrough (long-lived bearer token).
# No AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY required for the matrix --
# both the LiteLLM invoke and converse routes pick up
# AWS_BEARER_TOKEN_BEDROCK when present.
AWS_BEARER_TOKEN_BEDROCK=
AWS_REGION_NAME=us-east-1
# Vertex AI.
# On the GCP VM, the default service-account ADC from the metadata server
# is used -- no JSON key file is needed. If you ever need to run outside
# GCP, also export GOOGLE_APPLICATION_CREDENTIALS=/path/to/sa.json.
VERTEXAI_PROJECT=
VERTEXAI_LOCATION=global
# Microsoft Foundry (Azure column)
AZURE_FOUNDRY_API_KEY=
AZURE_FOUNDRY_API_BASE=
# REQUIRED for publishing: PAT for the `agent-shin` user, used to push
# the daily compat-matrix branch to its fork (agent-shin/litellm-docs)
# and open the cross-repo PR against BerriAI/litellm-docs. Scopes:
# classic `repo` + `workflow`, or fine-grained on agent-shin/litellm-docs
# with Contents:RW + Pull requests:RW + Workflows:RW.
# Skip by setting SKIP_PUBLISH=1 (publishes nothing; only writes the
# matrix JSON locally).
AGENT_SHIN_GITHUB_TOKEN=
# Optional: lifts the unauthenticated rate limit on the GitHub Releases
# API used by `resolver.py`. Any token works (read-only). Not required.
# GITHUB_TOKEN=
# Optional overrides; defaults are sensible for the cron VM.
# PROXY_PORT=4100
# LITELLM_WORKTREE=/home/mateo/litellm-cron-worktree
# DOCS_REPO=BerriAI/litellm-docs
# DOCS_BRANCH=main
# DOCS_TARGET_PATH=src/data/compatibility-matrix.json
# FORK_OWNER=agent-shin
# FORK_REPO=agent-shin/litellm-docs

View file

@ -0,0 +1,141 @@
# systemd service for the Claude Code compatibility-matrix populator.
#
# Triggered by `litellm-compat-matrix.timer`; not started directly. The
# unit is a `Type=oneshot` so the timer's `OnCalendar=` semantics
# describe "run once per day" cleanly — there's no long-lived daemon to
# supervise; each invocation runs the populator end-to-end and exits.
#
# Install
# -------
#
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service /etc/systemd/system/
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer /etc/systemd/system/
# sudo systemctl daemon-reload
# sudo systemctl enable --now litellm-compat-matrix.timer
#
# Paths are hard-coded to /home/mateo rather than using systemd's %h
# specifier. Why: in *system* units (this one), %h is expanded at
# parse time against the *manager's* home -- which is /root for PID 1
# -- and *not* against the User= directive. That mismatch makes
# ReadWritePaths point at /root/.cache (which doesn't exist), causing
# the namespace setup to fail with status=226/NAMESPACE before the
# script ever runs. The runtime user (`User=mateo`) must:
#
# * have a checkout of `BerriAI/litellm` at `~/litellm/litellm` so the
# publisher module is importable;
# * have a uv venv at `~/litellm/litellm/.venv` (created by
# `uv sync --frozen` inside that checkout once);
# * have `gh` already authenticated against an account with
# `pull-requests: write` on `BerriAI/litellm-docs`;
# * have provider credentials exported in `/etc/litellm-compat-matrix.env`
# (see `litellm-compat-matrix.env.example` in this directory).
[Unit]
Description=Claude Code compatibility-matrix populator (oneshot)
Wants=network-online.target
After=network-online.target
[Service]
Type=oneshot
User=mateo
Group=mateo
# Provider credentials + any gh/PROXY_PORT overrides live here. Format
# is the standard `KEY=value` one line per env var.
EnvironmentFile=-/etc/litellm-compat-matrix.env
# systemd starts with a minimal PATH (~/usr/local/bin:/usr/bin:/bin).
# `uv` and `claude` are installed under the runtime user's `~/.local/bin`
# so we have to prepend it explicitly; otherwise run_daily.sh fails at
# the up-front command-presence check.
Environment=PATH=/home/mateo/.local/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
# `HOME` is auto-set to /home/mateo when User=mateo is honored, but be
# explicit so anything that reads $HOME (e.g. uv's cache lookup, the
# claude CLI's per-session dir) sees the right value even if a future
# refactor flips DynamicUser= or PrivateUsers= on.
Environment=HOME=/home/mateo
WorkingDirectory=/home/mateo/litellm/litellm
ExecStart=/home/mateo/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
# 90 minutes is generous: cold runs do `git clone` + `uv sync` of a new
# tag's lockfile, which can take a couple of minutes on a 2-vCPU VM,
# plus 30 cells of pytest hitting four cloud providers.
TimeoutStartSec=90min
# A failed run shouldn't restart automatically — the next timer fire is
# the right retry. Reruns of the same day's matrix are idempotent.
Restart=no
# Security hardening: the populator only reads the litellm checkout and
# the env-file; everything else it writes lives in either the worktree
# (managed) or `/tmp` (cleaned up by tempfile).
#
# `ProtectHome=read-only` blocks writes to /home/mateo but still
# allows reads. That's safe for the trusted run_daily.sh script
# itself, but unsafe for any subprocess we don't control: a
# compromised npm-installed `claude` package, or a model-directed
# `Read` tool call during a PDF/vision cell, could read sensitive
# host files like `~/.config/gh/hosts.yml` (gh-host token),
# `~/.ssh/`, or `~/.bash_history`. We mitigate that at the call
# boundary: every `claude` subprocess (the up-front `claude --version`
# probe in run_daily.sh, plus every CLI invocation routed through
# tests/e2e/claude_code/cli_driver.py) runs with `HOME` pointed at a
# fresh empty per-invocation tmpdir, not at /home/mateo. The CLI
# never sees the runtime user's real dotfiles. `gh` invocations in
# run_daily.sh pass `GH_TOKEN` inline, so they never need to read
# ~/.config/gh either; that path is intentionally NOT in the
# whitelist below — keeping it out is the second line of defense if
# the inline-token convention is ever accidentally regressed.
#
# ReadWritePaths whitelist:
# * litellm-cron-worktree - the long-lived stable-tag checkout +
# its `.venv` (`uv sync` rewrites every
# run) + `.uv-bin` (pinned `uv` binary
# cache).
# * .cache - uv's wheel cache (~/.cache/uv) so we
# don't redownload pinned deps each
# run. Used only by the trusted `uv`
# process; not exposed to `claude`.
# * /tmp - mktemp -d workdir, proxy logs, and
# the per-`claude`-invocation isolated
# HOME tmpdirs. PrivateTmp=true below
# gives the service its own tmpfs view
# so these don't escape to the host.
NoNewPrivileges=true
ProtectSystem=strict
ProtectHome=read-only
ReadWritePaths=/home/mateo/litellm-cron-worktree /home/mateo/.cache /tmp
PrivateTmp=true
# Filesystem-level hiding for credential-bearing dotdirs/files. Even
# though `ProtectHome=read-only` prevents writes, a model-directed
# `Read` tool call (the PDF cells pass `--allowed-tools Read`) or a
# compromised `claude` package can read absolute paths under
# /home/mateo and exfiltrate the contents. `InaccessiblePaths=` makes
# the listed paths look like empty/missing to every process in the
# unit's mount namespace -- including the trusted populator script,
# which is fine because it doesn't need any of these:
#
# * .config/gh - gh CLI host token; we pass GH_TOKEN inline to
# every `gh` invocation (clone/PR/reviewer) so the
# host config is never consulted.
# * .ssh - never used by the populator.
# * .aws - upstream AWS credentials are passed to the proxy
# via the EnvironmentFile (provider env vars), not
# via shared SDK config files.
# * .docker - the populator never talks to a docker socket.
# * .kube - the populator never talks to a k8s API.
# * .gnupg - no GPG signing on the bot's commits.
#
# Leading `-` makes systemd tolerant if a path doesn't exist on the
# host (the unit is portable across VMs that may not have all of
# them set up). Anything else under /home/mateo (the litellm
# checkout, the cron worktree, the uv cache, .local/bin for the
# claude/uv/gh binaries on PATH) stays read-accessible.
InaccessiblePaths=-/home/mateo/.config/gh -/home/mateo/.ssh -/home/mateo/.aws -/home/mateo/.docker -/home/mateo/.kube -/home/mateo/.gnupg
[Install]
WantedBy=multi-user.target

View file

@ -0,0 +1,25 @@
# Daily timer for the compatibility-matrix populator.
#
# 06:00 UTC matches the original GitHub Actions cron schedule; chosen so
# operators in US/EU timezones see fresh PRs at the start of their work
# day.
#
# `Persistent=true` causes a missed run (VM was off / suspended) to
# fire the next time the timer is started, which is the property we
# want for a once-a-day job: the matrix should refresh as soon as the
# VM is reachable again, not wait another 24h.
#
# `RandomizedDelaySec=10min` smears load if multiple matrix-style
# pipelines are ever colocated on the same VM in the future.
[Unit]
Description=Run the Claude Code compatibility-matrix populator daily
[Timer]
OnCalendar=*-*-* 06:00:00 UTC
Persistent=true
RandomizedDelaySec=10min
Unit=litellm-compat-matrix.service
[Install]
WantedBy=timers.target

View file

@ -0,0 +1,590 @@
#!/usr/bin/env bash
# Daily Claude Code compatibility-matrix populator.
#
# Runs from the GCP VM `litellm-compatibility-matrix-populator` via the
# systemd timer in this directory. The flow is:
#
# 1. Resolve the latest LiteLLM v*-stable tag from the GitHub Releases API.
# 2. Update a long-lived worktree at $WORKTREE to that tag and `uv sync` it.
# 3. Boot the proxy as a background subprocess on $PROXY_PORT (default
# 4100; a separate port from the human-tended :4000 proxy).
# 4. Run `pytest tests/e2e/claude_code/` against the proxy. Test failures
# become `fail` cells in the JSON, not script errors.
# 5. Hand the per-test results artifact + manifest to a small Python
# CLI (`build_matrix.py`) that wraps the existing
# `matrix_builder.build_from_paths` to produce the published
# compatibility-matrix.json.
# 6. `gh repo clone` litellm-docs, write the JSON to a deterministic
# branch (`compat-matrix/<litellm>-<claude>-<UTC-date>`), commit,
# `git push --force`, and `gh pr create`.
#
# Same-day reruns land on the same branch so they update the existing PR
# rather than spawning a new one. If the JSON is byte-identical to the
# docs branch, we skip the push entirely.
#
# Required commands on $PATH: git, uv, gh, jq, curl, claude.
# Required state: ~/litellm/litellm checked out (this file lives in it),
# $WORKTREE is created on first run, gh is already authenticated.
#
# Override any default by setting the matching env var; see the systemd
# unit for the production wiring.
set -Eeuo pipefail
LITELLM_REPO="${LITELLM_REPO:-${HOME}/litellm/litellm}"
WORKTREE="${LITELLM_WORKTREE:-${HOME}/litellm-cron-worktree}"
PROXY_PORT="${PROXY_PORT:-4100}"
PROXY_API_KEY="${PROXY_API_KEY:-sk-cron-matrix}"
DOCS_REPO="${DOCS_REPO:-BerriAI/litellm-docs}"
DOCS_BRANCH="${DOCS_BRANCH:-main}"
DOCS_TARGET_PATH="${DOCS_TARGET_PATH:-src/data/compatibility-matrix.json}"
SKIP_PUBLISH="${SKIP_PUBLISH:-0}"
PYTEST_K="${PYTEST_K:-}"
# Comma-separated GitHub usernames to request a review from on every PR.
# Reviewers must have at least read access to ${DOCS_REPO}. PR-author
# (agent-shin) has implicit rights to request reviews from anyone with
# read access, so no extra token scope is needed. Set to empty to skip.
PR_REVIEWERS="${PR_REVIEWERS:-mateo-berri}"
POPULATOR_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
WORKDIR="$(mktemp -d -t litellm-compat-matrix.XXXXXX)"
PROXY_PID_FILE="${WORKDIR}/proxy.pid"
# Cleanup is intentionally aggressive: it can run on normal exit, on a
# signal received by the script, or after a partial failure where the
# proxy is up but ${PROXY_PID_FILE} is stale. We try four things in
# order and stop as soon as the proxy port is free:
#
# 1. SIGTERM the pid recorded in proxy.pid.
# 2. SIGKILL anything from `pgrep -f "litellm.*--port ${PROXY_PORT}"`
# that survived. This catches the common case where the recorded
# pid was the sh wrapper, not the long-lived python child.
# 3. ss -K on the port (kernel kills sockets but not processes;
# mostly useful for catching lingering CLOSE_WAITs).
# 4. wipe ${WORKDIR}.
cleanup() {
local rc=$?
set +e
local proxy_pid
if [[ -f "${PROXY_PID_FILE}" ]]; then
proxy_pid="$(cat "${PROXY_PID_FILE}")"
if [[ -n "${proxy_pid}" ]]; then
kill -TERM "-${proxy_pid}" 2>/dev/null || kill -TERM "${proxy_pid}" 2>/dev/null || true
for _ in 1 2 3 4 5; do
kill -0 "${proxy_pid}" 2>/dev/null || break
sleep 1
done
fi
fi
# Belt-and-braces: any python or uv talking to ${PROXY_PORT} that
# survived the SIGTERM gets SIGKILL'd by name.
pgrep -f "litellm.*--port[ =]?${PROXY_PORT}([^0-9]|$)" 2>/dev/null \
| xargs -r kill -KILL 2>/dev/null || true
pgrep -f "${WORKTREE}/.uv-bin/uv.*run litellm" 2>/dev/null \
| xargs -r kill -KILL 2>/dev/null || true
rm -rf "${WORKDIR}"
exit "${rc}"
}
trap cleanup EXIT INT TERM
log() { printf '==> %s\n' "$*" >&2; }
die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; }
for cmd in git uv gh jq curl claude; do
command -v "${cmd}" >/dev/null 2>&1 || die "missing required command: ${cmd}"
done
# Publishing is from a fork (agent-shin/litellm-docs) so neither the cron
# host nor the bot identity needs write access to BerriAI/litellm-docs. We
# require the fork token up front -- failing 30 minutes into a run because
# the env file is missing one line is a waste of CI quota.
if [[ "${SKIP_PUBLISH}" != "1" ]]; then
[[ -n "${AGENT_SHIN_GITHUB_TOKEN:-}" ]] \
|| die "AGENT_SHIN_GITHUB_TOKEN required to open PRs from agent-shin/litellm-docs (or set SKIP_PUBLISH=1)"
fi
# ---------------------------------------------------------------------------
# 1. Resolve versions
# ---------------------------------------------------------------------------
# Newest v*-stable release on BerriAI/litellm. The `select(...)` filter
# drops drafts/non-stable, the version_key sort handles 1.10 > 1.9.
#
# Paginate through the releases endpoint instead of grabbing only page 1
# (default page_size=30). LiteLLM ships multiple non-stable releases per
# day, so it's common to need to walk past 30+ entries before hitting
# the most recent v*-stable. We cap at 5 pages (500 releases) which is
# conservatively beyond the worst observed gap.
#
# We deliberately do NOT short-circuit on the first page that contains a
# v*-stable tag. The /releases endpoint orders by `created_at`, not by
# semver, so a backport on an older series (e.g. v1.80.1-stable cut
# today) can show up on an earlier page than a higher-versioned release
# (v1.83.0-stable cut two weeks ago). Breaking early on first-stable-seen
# would silently pin the cron to the stale tag because the
# higher-versioned release still on a later page would never make it
# into the merged set the `sort_by` below consumes. The only break we
# keep is the empty-page guard, which means a quiet period in the
# release feed doesn't waste API quota — we just always walk far enough
# to be confident we've seen the highest stable tag.
GH_AUTH_HEADER=()
if [[ -n "${GITHUB_TOKEN:-}" ]]; then
GH_AUTH_HEADER=(-H "Authorization: Bearer ${GITHUB_TOKEN}")
fi
RELEASES_JSON="${WORKDIR}/releases.json"
echo "[]" >"${RELEASES_JSON}"
for page in 1 2 3 4 5; do
PAGE_JSON="${WORKDIR}/releases.page${page}.json"
curl -fsS \
-H 'Accept: application/vnd.github+json' \
-H 'User-Agent: litellm-compat-matrix' \
"${GH_AUTH_HEADER[@]}" \
"https://api.github.com/repos/BerriAI/litellm/releases?per_page=100&page=${page}" \
>"${PAGE_JSON}"
jq -s '.[0] + .[1]' "${RELEASES_JSON}" "${PAGE_JSON}" >"${RELEASES_JSON}.merged"
mv "${RELEASES_JSON}.merged" "${RELEASES_JSON}"
# No more pages? GitHub returns an empty array past the last page.
if [[ "$(jq 'length' "${PAGE_JSON}")" == "0" ]]; then
break
fi
done
LITELLM_VERSION="$(
jq -r '
[ .[] | .tag_name // empty
| select(test("^v[0-9]+\\.[0-9]+\\.[0-9]+-stable$"))
]
| sort_by(
capture("^v(?<a>[0-9]+)\\.(?<b>[0-9]+)\\.(?<c>[0-9]+)-stable$")
| [(.a|tonumber), (.b|tonumber), (.c|tonumber)]
)
| last // empty
' "${RELEASES_JSON}"
)"
[[ -n "${LITELLM_VERSION}" ]] || die "could not resolve latest v*-stable tag in 5 pages of releases"
log "resolved litellm: ${LITELLM_VERSION}"
# The systemd unit loads provider credentials and the agent-shin GitHub
# token from /etc/litellm-compat-matrix.env into this script's
# environment. Running the npm-installed `claude` binary directly here
# would hand that full env to package code -- a compromised
# @anthropic-ai/claude-code release could read ANTHROPIC_API_KEY /
# AWS_BEARER_TOKEN_BEDROCK / AZURE_FOUNDRY_API_KEY /
# AGENT_SHIN_GITHUB_TOKEN from os.environ and exfiltrate them before
# the proxy or test harness ever starts. Probe under `env -i` with the
# same minimal allowlist the PR-gate uses (the matrix run itself goes
# through cli_driver.py, which already scrubs the CLI env).
#
# The probe also runs under a fresh empty HOME instead of the runtime
# user's real $HOME. `ProtectHome=read-only` in the systemd unit
# blocks *writes* to /home/mateo but still allows reads, so a
# compromised claude package invoked here with HOME=/home/mateo could
# read ~/.config/gh/hosts.yml (the gh-host token), ~/.bash_history,
# or ~/.ssh/. Pointing HOME at a per-run dir under ${WORKDIR} hides
# those entirely from the subprocess; ${WORKDIR} is rm -rf'd by the
# script-wide cleanup() trap regardless of probe outcome.
CLAUDE_PROBE_HOME="${WORKDIR}/claude-probe-home"
mkdir -p "${CLAUDE_PROBE_HOME}"
CLAUDE_CODE_VERSION="$(env -i \
PATH="${PATH}" \
HOME="${CLAUDE_PROBE_HOME}" \
USER="${USER:-mateo}" \
TERM="${TERM:-dumb}" \
LANG="${LANG:-C.UTF-8}" \
LC_ALL="${LC_ALL:-}" \
TMPDIR="${TMPDIR:-/tmp}" \
claude --version 2>/dev/null \
| grep -oE '[0-9]+\.[0-9]+\.[0-9]+([.-][A-Za-z0-9.-]+)?' \
| head -n1 || true)"
# `|| true` above keeps `set -Eeuo pipefail` from aborting silently when
# `grep` finds no match (exit 1) — without it the assignment inherits the
# pipeline's non-zero exit, `set -e` kills the script, and the operator
# never sees the helpful diagnostic below.
[[ -n "${CLAUDE_CODE_VERSION}" ]] || die "could not parse semver from 'claude --version'"
log "local claude code: ${CLAUDE_CODE_VERSION}"
# ---------------------------------------------------------------------------
# 2. Update the worktree to that tag
# ---------------------------------------------------------------------------
if [[ ! -d "${WORKTREE}/.git" ]]; then
log "first run: cloning litellm into ${WORKTREE}"
mkdir -p "$(dirname "${WORKTREE}")"
git clone https://github.com/BerriAI/litellm.git "${WORKTREE}"
fi
log "updating worktree to ${LITELLM_VERSION}"
git -C "${WORKTREE}" fetch --tags --force
git -C "${WORKTREE}" reset --hard
# Keep the venv and the .uv-bin cache around — uv sync will reconcile
# the venv on every run, and we don't want to re-download the pinned
# uv binary each time. Drop everything else (including any prior
# tests/e2e/claude_code/ shim) so each run starts clean before the shim
# below rewrites it from the dev checkout.
git -C "${WORKTREE}" clean -fdx -e .venv -e .uv-bin
git -C "${WORKTREE}" checkout --force "${LITELLM_VERSION}"
# Always overwrite tests/e2e/claude_code/ in the worktree with the copy
# from the dev checkout, regardless of whether the resolved
# ${LITELLM_VERSION} tag already ships a tests/e2e/claude_code/ tree of
# its own. Rationale: the matrix populator's job is to exercise
# today's tests against the latest stable proxy. The dev checkout
# carries the most recent test fixes (e.g. the stream-json vision
# rewrite, the --effort thinking knob, the WebSearch tool_use
# assertion) that haven't yet rolled into a v*-stable, and we want
# every cron run to pick those up the moment they land on
# ${LITELLM_REPO}, not whenever the next stable release happens.
#
# Concretely this means a fresh `rm -rf` + `cp -r` every run so the
# tree is byte-identical to ${LITELLM_REPO}/tests/e2e/claude_code (no
# stale files left over from the tag's own checkout, no drift across
# runs).
if [[ ! -d "${LITELLM_REPO}/tests/e2e/claude_code" ]]; then
die "no shim source at ${LITELLM_REPO}/tests/e2e/claude_code"
fi
log "shimming tests/e2e/claude_code/ from ${LITELLM_REPO} (always-overwrite)"
rm -rf "${WORKTREE}/tests/e2e/claude_code"
mkdir -p "${WORKTREE}/tests/e2e"
cp -r "${LITELLM_REPO}/tests/e2e/claude_code" "${WORKTREE}/tests/e2e/"
# litellm pins an exact uv version in pyproject.toml's [tool.uv]
# `required-version` field, so a system uv that's newer or older
# refuses to sync. We pin our own local copy at the version the
# checked-out tag asks for, cached under .uv-bin/ inside the worktree
# so subsequent runs skip the download.
PINNED_UV_VERSION="$(
awk -F'"' '
/^required-version[[:space:]]*=/ {
# Field 2 is the value between the quotes, e.g. ">=0.10.9" or
# "0.10.9". Strip any leading specifier prefix so we end up with
# the bare version string, which is what /releases/download/<v>/
# expects.
v = $2
sub(/^[[:space:]=<>!~]+/, "", v)
if (v != "") { print v; exit }
}
' "${WORKTREE}/pyproject.toml"
)"
if [[ -z "${PINNED_UV_VERSION}" ]]; then
log "no uv version pin in pyproject.toml; using system uv"
WORKTREE_UV="$(command -v uv)"
else
WORKTREE_UV="${WORKTREE}/.uv-bin/uv-${PINNED_UV_VERSION}"
if [[ ! -x "${WORKTREE_UV}" ]]; then
log "downloading uv ${PINNED_UV_VERSION} for the worktree"
mkdir -p "${WORKTREE}/.uv-bin"
# Detect host arch so the same script works on x86_64 GCP VMs and on
# aarch64 hosts (Astral publishes both `uv-x86_64-unknown-linux-gnu`
# and `uv-aarch64-unknown-linux-gnu` tarballs under the same release
# tag, and `uname -m` already returns the exact token uv uses).
UV_ARCH="$(uname -m)"
UV_TRIPLE="uv-${UV_ARCH}-unknown-linux-gnu"
UV_TARBALL_NAME="${UV_TRIPLE}.tar.gz"
UV_DOWNLOAD_URL="https://github.com/astral-sh/uv/releases/download/${PINNED_UV_VERSION}/${UV_TARBALL_NAME}"
UV_TMPDIR="$(mktemp -d -t uv-download.XXXXXX)"
# Download the tarball and Astral's official .sha256 sidecar to disk
# and verify the digest before extracting/executing anything. This
# closes the supply-chain trust gap of piping a remote binary
# straight into `tar -xzO ... > file ; chmod +x` (see CLAUDE.md
# "CI Supply-Chain Safety").
curl -fsSL --output "${UV_TMPDIR}/${UV_TARBALL_NAME}" "${UV_DOWNLOAD_URL}"
curl -fsSL --output "${UV_TMPDIR}/${UV_TARBALL_NAME}.sha256" "${UV_DOWNLOAD_URL}.sha256"
(cd "${UV_TMPDIR}" && sha256sum -c "${UV_TARBALL_NAME}.sha256") \
|| { rm -rf "${UV_TMPDIR}"; die "uv ${PINNED_UV_VERSION} sha256 mismatch — refusing to install"; }
tar -xzf "${UV_TMPDIR}/${UV_TARBALL_NAME}" -C "${UV_TMPDIR}" "${UV_TRIPLE}/uv"
mv "${UV_TMPDIR}/${UV_TRIPLE}/uv" "${WORKTREE_UV}.tmp"
chmod +x "${WORKTREE_UV}.tmp"
mv "${WORKTREE_UV}.tmp" "${WORKTREE_UV}"
rm -rf "${UV_TMPDIR}"
fi
fi
# `--extra proxy` pulls fastapi/uvicorn/etc. so `uv run litellm` can
# actually serve. `--group proxy-dev` brings in pytest and the rest of
# what tests/e2e/claude_code/ needs.
log "uv sync --frozen --group proxy-dev --extra proxy (uv ${PINNED_UV_VERSION:-system})"
(cd "${WORKTREE}" && "${WORKTREE_UV}" sync --frozen --group proxy-dev --extra proxy)
PROXY_CONFIG="${WORKTREE}/tests/e2e/claude_code/test_config.yaml"
[[ -f "${PROXY_CONFIG}" ]] || die "proxy config not found at ${PROXY_CONFIG} (does ${LITELLM_VERSION} predate the compat matrix work?)"
# ---------------------------------------------------------------------------
# 3. Boot the proxy
# ---------------------------------------------------------------------------
log "starting proxy on 127.0.0.1:${PROXY_PORT}"
# Bind the proxy to loopback only. The populator proxy is talked to
# exclusively by the pytest run on the same host (the health check and
# the test env set `LITELLM_PROXY_BASE_URL=http://127.0.0.1:...`),
# so there's no reason to expose it on the VM's external interfaces.
# Without `--host`, `litellm` defaults to 0.0.0.0, which combined with
# the predictable default `LITELLM_MASTER_KEY=sk-cron-matrix` would
# allow anything that can reach :${PROXY_PORT} on the VM to authenticate
# and burn upstream provider credentials.
#
# `setsid` puts the proxy in its own session+pgroup so cleanup() can
# SIGTERM the whole tree by passing the pgid as a negative pid. We
# write that pid to a file so cleanup() doesn't need to remember a
# variable that might be stale by the time the trap fires.
#
# Pass the master key as a shell-prefix assignment on `setsid` (inherited
# via the environment) rather than as `env KEY=VAL ...` argv. The argv
# form would land the literal key in /proc/<setsid-pid>/cmdline, where
# any local reader (a model-directed `Read` tool call, another user on
# the VM, a crash dump) could pick it up before the process execs into
# the litellm child. The shell-prefix form keeps the key out of argv at
# every layer (setsid → bash → uv → litellm).
LITELLM_MASTER_KEY="${PROXY_API_KEY}" setsid bash -c '
echo "$$" > "$0"
cd "$1"
exec "$2" run litellm --config "$3" --host 127.0.0.1 --port "$4"
' "${PROXY_PID_FILE}" "${WORKTREE}" "${WORKTREE_UV}" "${PROXY_CONFIG}" "${PROXY_PORT}" \
>"${WORKDIR}/proxy.log" 2>&1 &
disown
HEALTH_URL="http://127.0.0.1:${PROXY_PORT}/health/liveliness"
for _ in $(seq 1 45); do
if curl -fsS "${HEALTH_URL}" >/dev/null 2>&1; then
break
fi
sleep 2
done
curl -fsS "${HEALTH_URL}" >/dev/null \
|| { tail -50 "${WORKDIR}/proxy.log" >&2; die "proxy did not become healthy"; }
# ---------------------------------------------------------------------------
# 4. Run pytest
# ---------------------------------------------------------------------------
RESULTS_JSON="${WORKDIR}/compat-results.json"
PYTEST_ARGS=(
tests/e2e/claude_code/
--ignore=tests/e2e/claude_code/_driver_unit_tests
--ignore=tests/e2e/claude_code/_builder_unit_tests
--ignore=tests/e2e/claude_code/_publisher_unit_tests
--ignore=tests/e2e/claude_code/_pr_gate_unit_tests
)
if [[ -n "${PYTEST_K}" ]]; then
log "PYTEST_K set; narrowing to: ${PYTEST_K}"
PYTEST_ARGS+=(-k "${PYTEST_K}")
fi
log "running pytest"
set +e
# Pytest only needs to talk to the loopback proxy at 127.0.0.1:${PROXY_PORT}
# — it has no legitimate reason to see ANTHROPIC_API_KEY /
# AWS_BEARER_TOKEN_BEDROCK / VERTEXAI_* / AZURE_FOUNDRY_* /
# AGENT_SHIN_GITHUB_TOKEN / GITHUB_TOKEN in its own env. The systemd
# unit's EnvironmentFile injects all of those into this script for the
# proxy to consume, and pytest inherits them by default. Wrap the
# invocation in `env -i` so:
#
# 1. test code under tests/e2e/claude_code/ (or anything it imports)
# cannot read provider/agent-shin creds out of `os.environ` and
# exfiltrate them via an outbound call from inside a conftest hook
# or a fixture (a sibling vector to the model-controlled Bash/Read
# concern handled by `cli_driver.py`'s own env scrub);
# 2. a model-directed `Read` tool call during a PDF/vision cell
# cannot reach /proc/<pytest-pid>/environ and pull the creds out
# of the parent process the way it can today;
# 3. this matches the PR-gate pytest step in `.circleci/config.yml`,
# which already runs under `env -i` with the same minimal
# allowlist.
#
# `cli_driver.py` re-allowlists its own subset (PATH/USER/LOGNAME/etc.)
# when spawning the `claude` binary, so the CLI still finds Node + the
# claude shim on PATH and gets a fresh isolated HOME per invocation.
(
cd "${WORKTREE}" \
&& env -i \
PATH="${PATH}" \
HOME="${HOME}" \
USER="${USER:-mateo}" \
TERM="${TERM:-dumb}" \
LANG="${LANG:-C.UTF-8}" \
LC_ALL="${LC_ALL:-}" \
TMPDIR="${TMPDIR:-/tmp}" \
LITELLM_PROXY_BASE_URL="http://127.0.0.1:${PROXY_PORT}" \
LITELLM_PROXY_API_KEY="${PROXY_API_KEY}" \
COMPAT_RESULTS_PATH="${RESULTS_JSON}" \
"${WORKTREE_UV}" run pytest "${PYTEST_ARGS[@]}"
)
PYTEST_EXIT=$?
set -e
log "pytest exit code: ${PYTEST_EXIT} (failures become 'fail' cells, not script errors)"
[[ -f "${RESULTS_JSON}" ]] || die "pytest did not produce ${RESULTS_JSON}"
# ---------------------------------------------------------------------------
# 5. Build the matrix JSON
# ---------------------------------------------------------------------------
MATRIX_JSON="${WORKDIR}/compatibility-matrix.json"
log "building ${MATRIX_JSON}"
(
cd "${WORKTREE}" \
&& "${WORKTREE_UV}" run python "${POPULATOR_DIR}/build_matrix.py" \
--manifest "${WORKTREE}/tests/e2e/claude_code/manifest.yaml" \
--results "${RESULTS_JSON}" \
--output "${MATRIX_JSON}" \
--litellm-version "${LITELLM_VERSION}" \
--claude-code-version "${CLAUDE_CODE_VERSION}"
)
# ---------------------------------------------------------------------------
# 6. Open a docs-repo PR
# ---------------------------------------------------------------------------
if [[ "${SKIP_PUBLISH}" == "1" ]]; then
cp "${MATRIX_JSON}" "${LITELLM_REPO}/compatibility-matrix.json"
log "SKIP_PUBLISH=1; matrix written to ${LITELLM_REPO}/compatibility-matrix.json"
exit 0
fi
DATE_UTC="$(date -u +%Y-%m-%d)"
BRANCH_NAME="compat-matrix/${LITELLM_VERSION}-${CLAUDE_CODE_VERSION}-${DATE_UTC}"
DOCS_CLONE="${WORKDIR}/litellm-docs"
FORK_OWNER="${FORK_OWNER:-agent-shin}"
FORK_REPO="${FORK_REPO:-${FORK_OWNER}/litellm-docs}"
log "cloning ${DOCS_REPO}@${DOCS_BRANCH}"
# Use the agent-shin token inline rather than the host gh-cli config.
# `BerriAI/litellm-docs` is a public repo so unauthenticated clone
# would also work, but passing the token explicitly means the systemd
# unit can hide `~/.config/gh` (`InaccessiblePaths=`) without breaking
# this clone — closing the model-directed `Read("/home/mateo/.config/gh/...")`
# exfiltration path on the cron VM.
GH_TOKEN="${AGENT_SHIN_GITHUB_TOKEN}" \
gh repo clone "${DOCS_REPO}" "${DOCS_CLONE}" -- --depth 1 --branch "${DOCS_BRANCH}"
cd "${DOCS_CLONE}"
git config user.email "litellm-bot@berri.ai"
git config user.name "litellm-compat-matrix-bot"
git checkout -b "${BRANCH_NAME}"
mkdir -p "$(dirname "${DOCS_TARGET_PATH}")"
cp "${MATRIX_JSON}" "${DOCS_TARGET_PATH}"
git add "${DOCS_TARGET_PATH}"
if git diff --cached --quiet; then
log "matrix JSON unchanged from ${DOCS_BRANCH}; skipping PR"
exit 0
fi
GENERATED_AT="$(jq -r '.generated_at' "${MATRIX_JSON}")"
COMMIT_MSG="$(cat <<EOF
Update Claude Code compatibility matrix
litellm_version: ${LITELLM_VERSION}
claude_code_version: ${CLAUDE_CODE_VERSION}
generated_at: ${GENERATED_AT}
EOF
)"
git commit -m "${COMMIT_MSG}"
# Push to the fork (agent-shin/litellm-docs), not to BerriAI/litellm-docs.
# The cron host has no write access to BerriAI/litellm-docs by design --
# only agent-shin's PAT does, and only over its own fork. The temp remote
# carries the token in its URL, so we add it, push, then immediately
# remove it so the token never lingers in ${DOCS_CLONE}/.git/config.
# (${DOCS_CLONE} is also rm -rf'd by the cleanup trap on exit.)
#
# Plain --force (not --force-with-lease) is acceptable here: the fork
# branch is bot-owned, only this script ever writes to it, and runs are
# serialized by the systemd timer. --force-with-lease would require a
# fetch to populate the remote-tracking ref before each push and adds
# no safety in this single-writer setup.
FORK_PUSH_URL="https://x-access-token:${AGENT_SHIN_GITHUB_TOKEN}@github.com/${FORK_REPO}.git"
git remote remove fork 2>/dev/null || true
git remote add fork "${FORK_PUSH_URL}"
git push --force --set-upstream fork "${BRANCH_NAME}"
git remote remove fork
unset FORK_PUSH_URL
# Per-feature status table for the PR body. Reviewers triage from this.
PR_FEATURE_TABLE="$(jq -r '
.features[] as $f
| "- **\($f.name)**: " +
([ .providers[] as $p
| "\($p)=\($f.providers[$p].status // "not_tested")"
] | join(", "))
' "${MATRIX_JSON}")"
PR_TITLE="chore(compat-matrix): refresh for ${LITELLM_VERSION} + claude-code ${CLAUDE_CODE_VERSION}"
PR_BODY="$(cat <<EOF
Automated daily refresh of the Claude Code compatibility matrix.
| Field | Value |
| --- | --- |
| litellm_version | \`${LITELLM_VERSION}\` |
| claude_code_version | \`${CLAUDE_CODE_VERSION}\` |
| generated_at | \`${GENERATED_AT}\` |
## Per-feature results
${PR_FEATURE_TABLE}
---
Generated by \`tests/e2e/claude_code/cron_vm/run_daily.sh\`. Close without merging if the diff looks wrong; the next cron run will reopen with fresh results.
EOF
)"
log "opening PR from ${FORK_OWNER}:${BRANCH_NAME} -> ${DOCS_REPO}:${DOCS_BRANCH}"
# GH_TOKEN here is scoped to this single subshell so we don't bleed the
# fork token into the rest of the script (release-listing earlier uses
# ${GITHUB_TOKEN}, which may be a different identity). gh's --head accepts
# `OWNER:BRANCH` for cross-repo PRs from a fork.
#
# Reviewer assignment is done in a *separate* call below: as the PR
# author from a fork, agent-shin has no write/triage access on
# ${DOCS_REPO} and the `RequestReviewsByLogin` GraphQL mutation
# (which backs `gh pr create --reviewer` and `gh pr edit --add-reviewer`)
# rejects with "does not have the correct permissions". We use the
# collaborator-scoped ${GITHUB_TOKEN} for that instead. Don't fold
# --reviewer into `gh pr create` here -- it would fail the whole
# create on the very first cron run.
set +e
PR_OUT="$(
GH_TOKEN="${AGENT_SHIN_GITHUB_TOKEN}" gh pr create \
--repo "${DOCS_REPO}" \
--base "${DOCS_BRANCH}" \
--head "${FORK_OWNER}:${BRANCH_NAME}" \
--title "${PR_TITLE}" \
--body "${PR_BODY}" 2>&1
)"
PR_EXIT=$?
set -e
echo "${PR_OUT}"
if [[ ${PR_EXIT} -ne 0 ]]; then
if grep -q "a pull request for branch.*already exists" <<<"${PR_OUT}"; then
log "PR already exists for ${FORK_OWNER}:${BRANCH_NAME}; updated branch in place"
else
die "gh pr create failed (exit ${PR_EXIT})"
fi
fi
# Request reviews from PR_REVIEWERS using the collaborator-scoped
# ${GITHUB_TOKEN} (mateo-berri's token, already provisioned for release
# listing). This is idempotent: `gh pr edit --add-reviewer` is a no-op
# on a user who's already in reviewRequests, and silently re-adds
# anyone whose prior review was dismissed -- so same-day reruns stay
# clean. Reviewer-add failures are non-fatal: the matrix JSON has
# already landed on the PR; the worst case is a manual ping.
if [[ -n "${PR_REVIEWERS}" ]]; then
if [[ -z "${GITHUB_TOKEN:-}" ]]; then
log "WARN: PR_REVIEWERS set but GITHUB_TOKEN missing -- cannot request reviews; skipping"
else
log "requesting reviews from: ${PR_REVIEWERS}"
set +e
GH_TOKEN="${GITHUB_TOKEN}" gh pr edit \
"${FORK_OWNER}:${BRANCH_NAME}" \
--repo "${DOCS_REPO}" \
--add-reviewer "${PR_REVIEWERS}" 2>&1 | sed 's/^/ /'
REVIEWER_EXIT=${PIPESTATUS[0]}
set -e
if [[ ${REVIEWER_EXIT} -ne 0 ]]; then
log "WARN: gh pr edit --add-reviewer exited ${REVIEWER_EXIT} (non-fatal)"
fi
fi
fi
log "done"

View file

@ -0,0 +1,292 @@
"""Direct HTTP probe helpers for the Claude Code compatibility matrix.
Most matrix cells drive the `claude` CLI in headless mode and observe
the stream-json wire (see `cli_driver.py`). A handful of features the
proxy must support don't have any CLI surface area -- `count_tokens` is
the canonical example: Claude Code calls it internally for budget
display, but the result never appears in stream-json events, so a CLI
test cannot observe whether the endpoint round-tripped correctly
through the proxy for any given provider.
This module is the second test pattern the matrix supports: a plain
HTTP POST against a LiteLLM proxy endpoint, parsed and shape-checked
in the test, with the same `compat_result` recording convention as the
CLI-driven cells. The goal is to keep this pattern *narrow* -- if a
feature can be tested via the CLI, it should be, because the CLI path
is closer to what real Claude Code users hit. HTTP probes are only for
features the CLI can't reach.
The probe deliberately uses a short timeout (30s) and small payloads:
this is a "did the request shape survive the proxy's
provider-specific transformations" test, not a load test, and a real
endpoint regression typically surfaces in well under a second of wall
time (400 / 500 from the upstream, or LiteLLM 500 on a transformation
bug).
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from typing import Any, Mapping, Optional
import httpx
from claude_code.rate_limiter import (
RateLimiter,
get_default_limiter,
infer_provider,
)
DEFAULT_TIMEOUT_SECONDS = 30.0
@dataclass
class ProbeResult:
"""Structured outcome of a single HTTP probe.
`status_code` and `body` are the wire response; `payload` is the
parsed JSON body if the response was JSON, else None. Tests assert
on `status_code` + `payload` shape; `body` is preserved so failure
diagnostics can echo the raw error string (which is the only thing
a maintainer needs to triage a red cell).
"""
status_code: int
body: str
payload: Optional[Mapping[str, Any]] = None
error: Optional[str] = None
def probe_count_tokens(
*,
base_url: str,
api_key: str,
model: str,
message: str = "hello world",
timeout: float = DEFAULT_TIMEOUT_SECONDS,
rate_limiter: Optional[RateLimiter] = None,
) -> ProbeResult:
"""POST to `{base_url}/v1/messages/count_tokens` for `model` and return the parsed result.
The Anthropic / LiteLLM `count_tokens` endpoint accepts a request
body whose shape mirrors `/v1/messages` (model + messages), and
returns `{"input_tokens": N}` for a successful response. Anything
else -- non-200 status, non-JSON body, missing/non-int
`input_tokens` -- is a regression we want the cell to flip red on.
The same cross-process token-bucket limiter `cli_driver.run_claude`
uses is acquired here too, so probe rows count against the
aggregate per-provider budget. Without this, an HTTP-probe row
would fire unthrottled requests in parallel with throttled CLI
rows and silently violate the limiter's aggregate-rate guarantee.
`rate_limiter` is an injection seam for unit tests; production
callers should leave it unset to use the process-wide default.
"""
limiter = rate_limiter if rate_limiter is not None else get_default_limiter()
limiter.acquire(infer_provider(model))
url = base_url.rstrip("/") + "/v1/messages/count_tokens"
try:
response = httpx.post(
url,
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
# `anthropic-version` is required by Anthropic's native
# API and harmless on every other provider the proxy
# routes to. Matches what the Claude Code CLI sends
# for its own internal `count_tokens` calls.
"anthropic-version": "2023-06-01",
},
json={"model": model, "messages": [{"role": "user", "content": message}]},
timeout=timeout,
)
except httpx.HTTPError as exc:
return ProbeResult(status_code=0, body="", error=f"transport: {exc}")
body = response.text or ""
try:
payload = response.json() if body else None
except (json.JSONDecodeError, ValueError):
payload = None
return ProbeResult(
status_code=response.status_code,
body=body,
payload=payload,
)
def probe_tool_search(
*,
base_url: str,
api_key: str,
model: str,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
rate_limiter: Optional[RateLimiter] = None,
) -> ProbeResult:
"""POST to `{base_url}/v1/messages` with a `tool_search_tool_regex_20251119`
tool definition and return the result.
The shape of the tools array is the one Claude Code emits when its
MCP-tool-search beta is active: a `tool_search_tool_regex_20251119`
discovery tool (name `tool_search_tool_regex`) plus at least one
regular user tool to be searched. LiteLLM's
`is_tool_search_used` helper keys on the `_20251119`-suffixed type
string to decide whether to attach the provider-specific tool-search
beta header (`advanced-tool-use-2025-11-20` for Anthropic/Azure,
`tool-search-tool-2025-10-19` for Vertex/Bedrock). A proxy
regression in that translation will surface here as a 400 from
the upstream complaining about the tool type or beta header.
The prompt deliberately does not force a tool call -- the goal is
to verify the *request* round-trips without 400 and produces some
response, not to test whether the model decided to invoke
tool_search. That kind of behavior test would couple this row to
Claude Code's model behavior heuristics, which change weekly.
Like `probe_count_tokens`, this acquires one token from the
process-wide rate limiter so probe traffic counts against the
same aggregate per-provider budget as the CLI rows. `rate_limiter`
is a test seam; production callers should leave it unset.
"""
limiter = rate_limiter if rate_limiter is not None else get_default_limiter()
limiter.acquire(infer_provider(model))
url = base_url.rstrip("/") + "/v1/messages"
payload = {
"model": model,
"max_tokens": 64,
"messages": [
{
"role": "user",
"content": (
"If you have a tool to discover other tools, use it to "
"find one. Otherwise reply with the word 'done'."
),
}
],
"tools": [
# The tool_search discovery tool itself. Type is the SDK-
# version-pinned `_20251119` suffix; name is the canonical
# `tool_search_tool_regex` (no suffix) Anthropic accepts.
# LiteLLM keys its beta-header translation on the type.
{
"type": "tool_search_tool_regex_20251119",
"name": "tool_search_tool_regex",
},
# A trivial user tool for the discovery tool to potentially
# surface. Without at least one non-search tool the request
# is shape-valid but semantically empty; we include one so
# the wire shape mirrors what real Claude Code sends.
{
"name": "add_numbers",
"description": "Add two integers",
"input_schema": {
"type": "object",
"properties": {
"a": {"type": "integer"},
"b": {"type": "integer"},
},
"required": ["a", "b"],
},
},
],
}
try:
response = httpx.post(
url,
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
"anthropic-version": "2023-06-01",
},
json=payload,
timeout=timeout,
)
except httpx.HTTPError as exc:
return ProbeResult(status_code=0, body="", error=f"transport: {exc}")
body = response.text or ""
try:
payload_out = response.json() if body else None
except (json.JSONDecodeError, ValueError):
payload_out = None
return ProbeResult(
status_code=response.status_code,
body=body,
payload=payload_out,
)
def assert_tool_search_shape(result: ProbeResult) -> Optional[str]:
"""Return None on success, else describe the first violation.
Acceptance criteria:
1. HTTP status is 200 (no 400 from the upstream rejecting the
tool_search tool type or a missing beta header).
2. Body is valid JSON.
3. Body has either `content` (Anthropic-shape passthrough) or
`choices` (LiteLLM normalized openai-shape, used by Bedrock
Converse). Either is acceptable -- the matrix cares that the
proxy *accepts and forwards* tool_search, not that the model
actually chose to invoke it. Tool-invocation behavior is a
model decision the matrix has no business asserting on.
The cell goes red when the upstream rejects the tool type, the
proxy drops the beta header, or the response shape is unusable.
Anything else (model decided to call or not call tool_search) is
irrelevant for this row.
"""
if result.error is not None:
return f"transport error: {result.error}"
if result.status_code != 200:
return f"status {result.status_code}: {result.body[:400]}"
if result.payload is None:
return f"non-JSON body: {result.body[:400]}"
if not isinstance(result.payload, Mapping):
return f"body is not a JSON object: {type(result.payload).__name__}"
# LiteLLM normalizes some provider responses to OpenAI shape
# (`choices`) and passes others through Anthropic-shape (`content`).
# Accept either; both prove the proxy round-tripped the request.
if "content" not in result.payload and "choices" not in result.payload:
return (
f"response has neither `content` nor `choices`: "
f"keys={sorted(result.payload.keys())}"
)
return None
def assert_count_tokens_shape(result: ProbeResult) -> Optional[str]:
"""Return None on success, or an error string describing the first violation.
Acceptance criteria are intentionally minimal:
1. HTTP status is 200.
2. Body is valid JSON.
3. Body has an `input_tokens` key whose value is a positive int.
Anything beyond that (cache token fields, server metadata) is
optional and varies by provider/transport. Asserting on extras
would create a brittle test that flips red on neutral protocol
drift; matrix cells should only go red on functional regressions
a Claude Code user would feel.
"""
if result.error is not None:
return f"transport error: {result.error}"
if result.status_code != 200:
return f"status {result.status_code}: {result.body[:400]}"
if result.payload is None:
return f"non-JSON body: {result.body[:400]}"
if not isinstance(result.payload, Mapping):
return f"body is not a JSON object: {type(result.payload).__name__}"
tokens = result.payload.get("input_tokens")
if not isinstance(tokens, int) or isinstance(tokens, bool):
return f"input_tokens missing or not an int: got {tokens!r}"
if tokens <= 0:
return f"input_tokens must be positive; got {tokens}"
return None

View file

@ -0,0 +1,224 @@
"""long_context_1m x Anthropic.
Drive the real `claude` CLI in headless mode with a ~210k-token padded
prompt and the `--betas context-1m-2025-08-07` beta header, route
through a LiteLLM proxy aimed at Anthropic, and assert the request
round-trips: no 400 from a stripped beta header, no 413 from a body
the proxy refused to forward, and a non-empty assistant reply at the
end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/long_context_1m/test_anthropic.py
^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Cost note (read this before scaling the prompt up):
This row genuinely exercises the long-context path -- a 210k-token
prompt is *just over* Claude's standard 200k context window, which is
the threshold that requires the `context-1m-2025-08-07` beta header
to be honored end-to-end. Anything shorter would only test whether
the proxy forwards the beta header byte-for-byte; it would not catch
provider-side regressions where the header is forwarded but the
upstream silently truncates beyond the standard context (we've seen
this on third-party gateways). Anything longer is wasted spend.
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
Daily cost across all five providers (this row only): ~$19.
Haiku 4.5 is intentionally omitted: it does not support 1M context
(its window is 200k). Reporting `not_applicable` for Haiku would
flip the entire cell to `not_applicable`, hiding genuine 1M
regressions on Sonnet/Opus; instead we exclude Haiku from the model
list entirely and let the matrix's per-cell aggregator green the
cell on Sonnet + Opus passing. This is the one row where the "all
three tiers must pass" rule is relaxed; it's relaxed structurally
(via the model list), not semantically (via not_applicable), so the
matrix builder stays unmodified.
The prompt is delivered via subprocess stdin rather than a positional
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
fits within that comfortably, but stdin is safer (no shell escaping
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
keeps the driver's `extra_args` slot free for the `--betas` flag.
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
that loops three Claude tiers in a single cell can't accidentally
spend more than ~$18 on this cell. The cap is twice the expected
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
provider's pricing changes and the matrix starts spending more than
$10/day on this row.
"""
from __future__ import annotations
import os
from typing import Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
# 1M-context beta. See module docstring for the per-cell-aggregator
# rationale.
ANTHROPIC_MODELS: Sequence[str] = (
"claude-sonnet-4-6",
"claude-opus-4-7",
)
# Beta header that opts an Anthropic-shape model into the 1M context
# window. Same string is accepted on Bedrock (Invoke + Converse) and
# Vertex per LiteLLM's transformers -- no per-provider translation is
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
# `tool-search-tool-2025-10-19`.
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
# Target a padded prompt that lands just above Claude's standard 200k
# context window so the request can only succeed if the
# `context-1m-2025-08-07` beta header survives all the way to the
# upstream. Below 200k the cell would silently pass even with a
# proxy-dropped beta header; above ~220k we're paying for tokens that
# don't add signal.
TARGET_INPUT_TOKENS = 210_000
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
# through several benign pangrams + filler so the padding looks like a
# real document, not a repeating monolith. Identical-line padding +
# "ignore everything above" trips Opus 4.7's safety filter as a
# suspected prompt-injection attempt -- we hit that during smoke
# testing and the cell flipped red for the wrong reason. Varied prose
# with a natural document-style framing keeps the filter quiet.
_PAD_CHUNKS = (
"The quick brown fox jumps over the lazy dog. ",
"She sells seashells by the seashore on Sunday mornings. ",
"Pack my box with five dozen liquor jugs for the journey. ",
"How vexingly quick daft zebras jump over fences at dawn. ",
"Sphinx of black quartz, judge my vow of silence and patience. ",
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
)
_CHARS_PER_TOKEN = 4
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
Framing:
- Lead with a benign document-style preamble that justifies the
long context (so safety filters see the prompt as "long
document review" rather than "adversarial padding").
- Cycle through a small set of pangrams + filler sentences for
the bulk of the padding. Variety matters: identical repeated
lines look like a denial-of-service or injection attempt to
Anthropic's content filter on the larger tiers.
- End with the actual question. Claude's instruction-following
is stronger on recent tokens, so a 210k-token-into-the-past
instruction would risk a false-fail where the model ignores
it.
`target_tokens` is an approximation: actual token count depends
on the tokenizer, but Anthropic's English tokenizer averages
~4 chars/token, so 4 × target_tokens chars of padding gets us
close enough to the 1M-beta threshold (200k) that the proxy's
beta-header handling is the only path to success.
"""
preamble = (
"I'm going to share an excerpt from a long document with you. "
"It contains a mix of practice sentences a typist might use to "
"warm up; treat the bulk of the text as background context. "
"I'll ask a short question at the end.\n\n"
"Begin excerpt:\n\n"
)
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
pad_lines = []
pad_len = 0
idx = 0
while pad_len < pad_target_chars:
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
pad_lines.append(chunk)
pad_len += len(chunk)
idx += 1
return preamble + "".join(pad_lines) + closing
def test_long_context_1m_anthropic(compat_result):
"""Drive the `claude` CLI with a ~210k-token prompt and the
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
non-empty reply for Sonnet + Opus."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
long_prompt = _build_long_prompt()
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt=None,
stdin_input=long_prompt,
base_url=base_url,
api_key=api_key,
extra_args=[
"--betas",
LONG_CONTEXT_BETA,
# Hard ceiling so a runaway test cannot blow the budget.
# See module docstring for sizing.
"--max-budget-usd",
"6",
],
# Long-context requests can take a couple of minutes on a
# loaded upstream; the driver's default 120s is too tight.
timeout=300.0,
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,224 @@
"""long_context_1m x Azure (Microsoft Foundry).
Drive the real `claude` CLI in headless mode with a ~210k-token padded
prompt and the `--betas context-1m-2025-08-07` beta header, route
through a LiteLLM proxy aimed at Anthropic, and assert the request
round-trips: no 400 from a stripped beta header, no 413 from a body
the proxy refused to forward, and a non-empty assistant reply at the
end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/long_context_1m/test_azure.py
^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
Cost note (read this before scaling the prompt up):
This row genuinely exercises the long-context path -- a 210k-token
prompt is *just over* Claude's standard 200k context window, which is
the threshold that requires the `context-1m-2025-08-07` beta header
to be honored end-to-end. Anything shorter would only test whether
the proxy forwards the beta header byte-for-byte; it would not catch
provider-side regressions where the header is forwarded but the
upstream silently truncates beyond the standard context (we've seen
this on third-party gateways). Anything longer is wasted spend.
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
Daily cost across all five providers (this row only): ~$19.
Haiku 4.5 is intentionally omitted: it does not support 1M context
(its window is 200k). Reporting `not_applicable` for Haiku would
flip the entire cell to `not_applicable`, hiding genuine 1M
regressions on Sonnet/Opus; instead we exclude Haiku from the model
list entirely and let the matrix's per-cell aggregator green the
cell on Sonnet + Opus passing. This is the one row where the "all
three tiers must pass" rule is relaxed; it's relaxed structurally
(via the model list), not semantically (via not_applicable), so the
matrix builder stays unmodified.
The prompt is delivered via subprocess stdin rather than a positional
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
fits within that comfortably, but stdin is safer (no shell escaping
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
keeps the driver's `extra_args` slot free for the `--betas` flag.
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
that loops three Claude tiers in a single cell can't accidentally
spend more than ~$18 on this cell. The cap is twice the expected
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
provider's pricing changes and the matrix starts spending more than
$10/day on this row.
"""
from __future__ import annotations
import os
from typing import Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
# 1M-context beta. See module docstring for the per-cell-aggregator
# rationale.
AZURE_MODELS: Sequence[str] = (
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
)
# Beta header that opts an Anthropic-shape model into the 1M context
# window. Same string is accepted on Bedrock (Invoke + Converse) and
# Vertex per LiteLLM's transformers -- no per-provider translation is
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
# `tool-search-tool-2025-10-19`.
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
# Target a padded prompt that lands just above Claude's standard 200k
# context window so the request can only succeed if the
# `context-1m-2025-08-07` beta header survives all the way to the
# upstream. Below 200k the cell would silently pass even with a
# proxy-dropped beta header; above ~220k we're paying for tokens that
# don't add signal.
TARGET_INPUT_TOKENS = 210_000
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
# through several benign pangrams + filler so the padding looks like a
# real document, not a repeating monolith. Identical-line padding +
# "ignore everything above" trips Opus 4.7's safety filter as a
# suspected prompt-injection attempt -- we hit that during smoke
# testing and the cell flipped red for the wrong reason. Varied prose
# with a natural document-style framing keeps the filter quiet.
_PAD_CHUNKS = (
"The quick brown fox jumps over the lazy dog. ",
"She sells seashells by the seashore on Sunday mornings. ",
"Pack my box with five dozen liquor jugs for the journey. ",
"How vexingly quick daft zebras jump over fences at dawn. ",
"Sphinx of black quartz, judge my vow of silence and patience. ",
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
)
_CHARS_PER_TOKEN = 4
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
Framing:
- Lead with a benign document-style preamble that justifies the
long context (so safety filters see the prompt as "long
document review" rather than "adversarial padding").
- Cycle through a small set of pangrams + filler sentences for
the bulk of the padding. Variety matters: identical repeated
lines look like a denial-of-service or injection attempt to
Anthropic's content filter on the larger tiers.
- End with the actual question. Claude's instruction-following
is stronger on recent tokens, so a 210k-token-into-the-past
instruction would risk a false-fail where the model ignores
it.
`target_tokens` is an approximation: actual token count depends
on the tokenizer, but Anthropic's English tokenizer averages
~4 chars/token, so 4 × target_tokens chars of padding gets us
close enough to the 1M-beta threshold (200k) that the proxy's
beta-header handling is the only path to success.
"""
preamble = (
"I'm going to share an excerpt from a long document with you. "
"It contains a mix of practice sentences a typist might use to "
"warm up; treat the bulk of the text as background context. "
"I'll ask a short question at the end.\n\n"
"Begin excerpt:\n\n"
)
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
pad_lines = []
pad_len = 0
idx = 0
while pad_len < pad_target_chars:
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
pad_lines.append(chunk)
pad_len += len(chunk)
idx += 1
return preamble + "".join(pad_lines) + closing
def test_long_context_1m_azure(compat_result):
"""Drive the `claude` CLI (Azure (Microsoft Foundry)) with a ~210k-token prompt and the
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
non-empty reply for Sonnet + Opus."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
long_prompt = _build_long_prompt()
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=None,
stdin_input=long_prompt,
base_url=base_url,
api_key=api_key,
extra_args=[
"--betas",
LONG_CONTEXT_BETA,
# Hard ceiling so a runaway test cannot blow the budget.
# See module docstring for sizing.
"--max-budget-usd",
"6",
],
# Long-context requests can take a couple of minutes on a
# loaded upstream; the driver's default 120s is too tight.
timeout=300.0,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,224 @@
"""long_context_1m x Bedrock (Converse).
Drive the real `claude` CLI in headless mode with a ~210k-token padded
prompt and the `--betas context-1m-2025-08-07` beta header, route
through a LiteLLM proxy aimed at Anthropic, and assert the request
round-trips: no 400 from a stripped beta header, no 413 from a body
the proxy refused to forward, and a non-empty assistant reply at the
end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py
^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
Cost note (read this before scaling the prompt up):
This row genuinely exercises the long-context path -- a 210k-token
prompt is *just over* Claude's standard 200k context window, which is
the threshold that requires the `context-1m-2025-08-07` beta header
to be honored end-to-end. Anything shorter would only test whether
the proxy forwards the beta header byte-for-byte; it would not catch
provider-side regressions where the header is forwarded but the
upstream silently truncates beyond the standard context (we've seen
this on third-party gateways). Anything longer is wasted spend.
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
Daily cost across all five providers (this row only): ~$19.
Haiku 4.5 is intentionally omitted: it does not support 1M context
(its window is 200k). Reporting `not_applicable` for Haiku would
flip the entire cell to `not_applicable`, hiding genuine 1M
regressions on Sonnet/Opus; instead we exclude Haiku from the model
list entirely and let the matrix's per-cell aggregator green the
cell on Sonnet + Opus passing. This is the one row where the "all
three tiers must pass" rule is relaxed; it's relaxed structurally
(via the model list), not semantically (via not_applicable), so the
matrix builder stays unmodified.
The prompt is delivered via subprocess stdin rather than a positional
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
fits within that comfortably, but stdin is safer (no shell escaping
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
keeps the driver's `extra_args` slot free for the `--betas` flag.
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
that loops three Claude tiers in a single cell can't accidentally
spend more than ~$18 on this cell. The cap is twice the expected
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
provider's pricing changes and the matrix starts spending more than
$10/day on this row.
"""
from __future__ import annotations
import os
from typing import Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
# 1M-context beta. See module docstring for the per-cell-aggregator
# rationale.
BEDROCK_CONVERSE_MODELS: Sequence[str] = (
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
)
# Beta header that opts an Anthropic-shape model into the 1M context
# window. Same string is accepted on Bedrock (Invoke + Converse) and
# Vertex per LiteLLM's transformers -- no per-provider translation is
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
# `tool-search-tool-2025-10-19`.
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
# Target a padded prompt that lands just above Claude's standard 200k
# context window so the request can only succeed if the
# `context-1m-2025-08-07` beta header survives all the way to the
# upstream. Below 200k the cell would silently pass even with a
# proxy-dropped beta header; above ~220k we're paying for tokens that
# don't add signal.
TARGET_INPUT_TOKENS = 210_000
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
# through several benign pangrams + filler so the padding looks like a
# real document, not a repeating monolith. Identical-line padding +
# "ignore everything above" trips Opus 4.7's safety filter as a
# suspected prompt-injection attempt -- we hit that during smoke
# testing and the cell flipped red for the wrong reason. Varied prose
# with a natural document-style framing keeps the filter quiet.
_PAD_CHUNKS = (
"The quick brown fox jumps over the lazy dog. ",
"She sells seashells by the seashore on Sunday mornings. ",
"Pack my box with five dozen liquor jugs for the journey. ",
"How vexingly quick daft zebras jump over fences at dawn. ",
"Sphinx of black quartz, judge my vow of silence and patience. ",
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
)
_CHARS_PER_TOKEN = 4
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
Framing:
- Lead with a benign document-style preamble that justifies the
long context (so safety filters see the prompt as "long
document review" rather than "adversarial padding").
- Cycle through a small set of pangrams + filler sentences for
the bulk of the padding. Variety matters: identical repeated
lines look like a denial-of-service or injection attempt to
Anthropic's content filter on the larger tiers.
- End with the actual question. Claude's instruction-following
is stronger on recent tokens, so a 210k-token-into-the-past
instruction would risk a false-fail where the model ignores
it.
`target_tokens` is an approximation: actual token count depends
on the tokenizer, but Anthropic's English tokenizer averages
~4 chars/token, so 4 × target_tokens chars of padding gets us
close enough to the 1M-beta threshold (200k) that the proxy's
beta-header handling is the only path to success.
"""
preamble = (
"I'm going to share an excerpt from a long document with you. "
"It contains a mix of practice sentences a typist might use to "
"warm up; treat the bulk of the text as background context. "
"I'll ask a short question at the end.\n\n"
"Begin excerpt:\n\n"
)
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
pad_lines = []
pad_len = 0
idx = 0
while pad_len < pad_target_chars:
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
pad_lines.append(chunk)
pad_len += len(chunk)
idx += 1
return preamble + "".join(pad_lines) + closing
def test_long_context_1m_bedrock_converse(compat_result):
"""Drive the `claude` CLI (Bedrock (Converse)) with a ~210k-token prompt and the
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
non-empty reply for Sonnet + Opus."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
long_prompt = _build_long_prompt()
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt=None,
stdin_input=long_prompt,
base_url=base_url,
api_key=api_key,
extra_args=[
"--betas",
LONG_CONTEXT_BETA,
# Hard ceiling so a runaway test cannot blow the budget.
# See module docstring for sizing.
"--max-budget-usd",
"6",
],
# Long-context requests can take a couple of minutes on a
# loaded upstream; the driver's default 120s is too tight.
timeout=300.0,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,224 @@
"""long_context_1m x Bedrock (Invoke).
Drive the real `claude` CLI in headless mode with a ~210k-token padded
prompt and the `--betas context-1m-2025-08-07` beta header, route
through a LiteLLM proxy aimed at Anthropic, and assert the request
round-trips: no 400 from a stripped beta header, no 413 from a body
the proxy refused to forward, and a non-empty assistant reply at the
end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py
^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
Cost note (read this before scaling the prompt up):
This row genuinely exercises the long-context path -- a 210k-token
prompt is *just over* Claude's standard 200k context window, which is
the threshold that requires the `context-1m-2025-08-07` beta header
to be honored end-to-end. Anything shorter would only test whether
the proxy forwards the beta header byte-for-byte; it would not catch
provider-side regressions where the header is forwarded but the
upstream silently truncates beyond the standard context (we've seen
this on third-party gateways). Anything longer is wasted spend.
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
Daily cost across all five providers (this row only): ~$19.
Haiku 4.5 is intentionally omitted: it does not support 1M context
(its window is 200k). Reporting `not_applicable` for Haiku would
flip the entire cell to `not_applicable`, hiding genuine 1M
regressions on Sonnet/Opus; instead we exclude Haiku from the model
list entirely and let the matrix's per-cell aggregator green the
cell on Sonnet + Opus passing. This is the one row where the "all
three tiers must pass" rule is relaxed; it's relaxed structurally
(via the model list), not semantically (via not_applicable), so the
matrix builder stays unmodified.
The prompt is delivered via subprocess stdin rather than a positional
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
fits within that comfortably, but stdin is safer (no shell escaping
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
keeps the driver's `extra_args` slot free for the `--betas` flag.
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
that loops three Claude tiers in a single cell can't accidentally
spend more than ~$18 on this cell. The cap is twice the expected
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
provider's pricing changes and the matrix starts spending more than
$10/day on this row.
"""
from __future__ import annotations
import os
from typing import Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
# 1M-context beta. See module docstring for the per-cell-aggregator
# rationale.
BEDROCK_INVOKE_MODELS: Sequence[str] = (
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
)
# Beta header that opts an Anthropic-shape model into the 1M context
# window. Same string is accepted on Bedrock (Invoke + Converse) and
# Vertex per LiteLLM's transformers -- no per-provider translation is
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
# `tool-search-tool-2025-10-19`.
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
# Target a padded prompt that lands just above Claude's standard 200k
# context window so the request can only succeed if the
# `context-1m-2025-08-07` beta header survives all the way to the
# upstream. Below 200k the cell would silently pass even with a
# proxy-dropped beta header; above ~220k we're paying for tokens that
# don't add signal.
TARGET_INPUT_TOKENS = 210_000
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
# through several benign pangrams + filler so the padding looks like a
# real document, not a repeating monolith. Identical-line padding +
# "ignore everything above" trips Opus 4.7's safety filter as a
# suspected prompt-injection attempt -- we hit that during smoke
# testing and the cell flipped red for the wrong reason. Varied prose
# with a natural document-style framing keeps the filter quiet.
_PAD_CHUNKS = (
"The quick brown fox jumps over the lazy dog. ",
"She sells seashells by the seashore on Sunday mornings. ",
"Pack my box with five dozen liquor jugs for the journey. ",
"How vexingly quick daft zebras jump over fences at dawn. ",
"Sphinx of black quartz, judge my vow of silence and patience. ",
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
)
_CHARS_PER_TOKEN = 4
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
Framing:
- Lead with a benign document-style preamble that justifies the
long context (so safety filters see the prompt as "long
document review" rather than "adversarial padding").
- Cycle through a small set of pangrams + filler sentences for
the bulk of the padding. Variety matters: identical repeated
lines look like a denial-of-service or injection attempt to
Anthropic's content filter on the larger tiers.
- End with the actual question. Claude's instruction-following
is stronger on recent tokens, so a 210k-token-into-the-past
instruction would risk a false-fail where the model ignores
it.
`target_tokens` is an approximation: actual token count depends
on the tokenizer, but Anthropic's English tokenizer averages
~4 chars/token, so 4 × target_tokens chars of padding gets us
close enough to the 1M-beta threshold (200k) that the proxy's
beta-header handling is the only path to success.
"""
preamble = (
"I'm going to share an excerpt from a long document with you. "
"It contains a mix of practice sentences a typist might use to "
"warm up; treat the bulk of the text as background context. "
"I'll ask a short question at the end.\n\n"
"Begin excerpt:\n\n"
)
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
pad_lines = []
pad_len = 0
idx = 0
while pad_len < pad_target_chars:
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
pad_lines.append(chunk)
pad_len += len(chunk)
idx += 1
return preamble + "".join(pad_lines) + closing
def test_long_context_1m_bedrock_invoke(compat_result):
"""Drive the `claude` CLI (Bedrock (Invoke)) with a ~210k-token prompt and the
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
non-empty reply for Sonnet + Opus."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
long_prompt = _build_long_prompt()
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=None,
stdin_input=long_prompt,
base_url=base_url,
api_key=api_key,
extra_args=[
"--betas",
LONG_CONTEXT_BETA,
# Hard ceiling so a runaway test cannot blow the budget.
# See module docstring for sizing.
"--max-budget-usd",
"6",
],
# Long-context requests can take a couple of minutes on a
# loaded upstream; the driver's default 120s is too tight.
timeout=300.0,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,224 @@
"""long_context_1m x Vertex AI.
Drive the real `claude` CLI in headless mode with a ~210k-token padded
prompt and the `--betas context-1m-2025-08-07` beta header, route
through a LiteLLM proxy aimed at Anthropic, and assert the request
round-trips: no 400 from a stripped beta header, no 413 from a body
the proxy refused to forward, and a non-empty assistant reply at the
end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/long_context_1m/test_vertex_ai.py
^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
Cost note (read this before scaling the prompt up):
This row genuinely exercises the long-context path -- a 210k-token
prompt is *just over* Claude's standard 200k context window, which is
the threshold that requires the `context-1m-2025-08-07` beta header
to be honored end-to-end. Anything shorter would only test whether
the proxy forwards the beta header byte-for-byte; it would not catch
provider-side regressions where the header is forwarded but the
upstream silently truncates beyond the standard context (we've seen
this on third-party gateways). Anything longer is wasted spend.
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
Daily cost across all five providers (this row only): ~$19.
Haiku 4.5 is intentionally omitted: it does not support 1M context
(its window is 200k). Reporting `not_applicable` for Haiku would
flip the entire cell to `not_applicable`, hiding genuine 1M
regressions on Sonnet/Opus; instead we exclude Haiku from the model
list entirely and let the matrix's per-cell aggregator green the
cell on Sonnet + Opus passing. This is the one row where the "all
three tiers must pass" rule is relaxed; it's relaxed structurally
(via the model list), not semantically (via not_applicable), so the
matrix builder stays unmodified.
The prompt is delivered via subprocess stdin rather than a positional
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
fits within that comfortably, but stdin is safer (no shell escaping
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
keeps the driver's `extra_args` slot free for the `--betas` flag.
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
that loops three Claude tiers in a single cell can't accidentally
spend more than ~$18 on this cell. The cap is twice the expected
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
provider's pricing changes and the matrix starts spending more than
$10/day on this row.
"""
from __future__ import annotations
import os
from typing import Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
# 1M-context beta. See module docstring for the per-cell-aggregator
# rationale.
VERTEX_AI_MODELS: Sequence[str] = (
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
)
# Beta header that opts an Anthropic-shape model into the 1M context
# window. Same string is accepted on Bedrock (Invoke + Converse) and
# Vertex per LiteLLM's transformers -- no per-provider translation is
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
# `tool-search-tool-2025-10-19`.
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
# Target a padded prompt that lands just above Claude's standard 200k
# context window so the request can only succeed if the
# `context-1m-2025-08-07` beta header survives all the way to the
# upstream. Below 200k the cell would silently pass even with a
# proxy-dropped beta header; above ~220k we're paying for tokens that
# don't add signal.
TARGET_INPUT_TOKENS = 210_000
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
# through several benign pangrams + filler so the padding looks like a
# real document, not a repeating monolith. Identical-line padding +
# "ignore everything above" trips Opus 4.7's safety filter as a
# suspected prompt-injection attempt -- we hit that during smoke
# testing and the cell flipped red for the wrong reason. Varied prose
# with a natural document-style framing keeps the filter quiet.
_PAD_CHUNKS = (
"The quick brown fox jumps over the lazy dog. ",
"She sells seashells by the seashore on Sunday mornings. ",
"Pack my box with five dozen liquor jugs for the journey. ",
"How vexingly quick daft zebras jump over fences at dawn. ",
"Sphinx of black quartz, judge my vow of silence and patience. ",
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
)
_CHARS_PER_TOKEN = 4
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
Framing:
- Lead with a benign document-style preamble that justifies the
long context (so safety filters see the prompt as "long
document review" rather than "adversarial padding").
- Cycle through a small set of pangrams + filler sentences for
the bulk of the padding. Variety matters: identical repeated
lines look like a denial-of-service or injection attempt to
Anthropic's content filter on the larger tiers.
- End with the actual question. Claude's instruction-following
is stronger on recent tokens, so a 210k-token-into-the-past
instruction would risk a false-fail where the model ignores
it.
`target_tokens` is an approximation: actual token count depends
on the tokenizer, but Anthropic's English tokenizer averages
~4 chars/token, so 4 × target_tokens chars of padding gets us
close enough to the 1M-beta threshold (200k) that the proxy's
beta-header handling is the only path to success.
"""
preamble = (
"I'm going to share an excerpt from a long document with you. "
"It contains a mix of practice sentences a typist might use to "
"warm up; treat the bulk of the text as background context. "
"I'll ask a short question at the end.\n\n"
"Begin excerpt:\n\n"
)
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
pad_lines = []
pad_len = 0
idx = 0
while pad_len < pad_target_chars:
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
pad_lines.append(chunk)
pad_len += len(chunk)
idx += 1
return preamble + "".join(pad_lines) + closing
def test_long_context_1m_vertex_ai(compat_result):
"""Drive the `claude` CLI (Vertex AI) with a ~210k-token prompt and the
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
non-empty reply for Sonnet + Opus."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
long_prompt = _build_long_prompt()
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt=None,
stdin_input=long_prompt,
base_url=base_url,
api_key=api_key,
extra_args=[
"--betas",
LONG_CONTEXT_BETA,
# Hard ceiling so a runaway test cannot blow the budget.
# See module docstring for sizing.
"--max-budget-usd",
"6",
],
# Long-context requests can take a couple of minutes on a
# loaded upstream; the driver's default 120s is too tight.
timeout=300.0,
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not outcome.text.strip():
error = f"[{model}] claude returned empty assistant text"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,103 @@
# Claude Code Compatibility Matrix — feature manifest.
#
# Defines the row order of the matrix and maps each feature_id to its
# human-readable display name. Adding a new feature to the matrix is a
# three-step change:
# 1. Append an entry to `features:` below.
# 2. Create a directory `tests/e2e/claude_code/<feature_id>/`.
# 3. Add per-provider test files inside that directory.
#
# `feature_id` MUST match the directory name on disk; the test harness
# infers (feature, provider) for each test from its file path.
schema_version: "1"
# Provider column order in the rendered matrix.
providers:
- anthropic
- bedrock_invoke
- bedrock_converse
- vertex_ai
- azure
# Feature row order.
features:
- id: basic_messaging_non_streaming
name: Basic messaging (non-streaming)
- id: basic_messaging_streaming
name: Basic messaging (streaming)
- id: tool_use
name: Tool use
- id: prompt_caching_5m
name: Prompt caching (5m TTL)
- id: vision
name: Vision
- id: thinking
name: Thinking
# The single row covers both API shapes Anthropic exposes — manual
# `thinking: {type: "enabled", budget_tokens: N}` (Haiku 4.5) and
# `thinking: {type: "adaptive"}` (Opus 4.7); Sonnet 4.6 supports
# either and Claude Code picks per model. A break in either
# transformer surfaces as a red cell because all three tiers must
# pass for the cell to go green. The row was named
# `extended_thinking` historically; Anthropic's docs now reserve
# that name for the deprecated manual mode only, so the row was
# renamed to the feature-level "Thinking".
- id: tool_use_streaming
name: Tool use (streaming / fine-grained)
- id: thinking_with_tool_use
name: Extended thinking + tool use
- id: pdf_input
name: PDF document input
- id: prompt_caching_1h
name: Prompt caching (1h TTL)
- id: web_search
name: Web search (server tool)
- id: structured_outputs
name: Structured outputs
# Drives `claude --json-schema '<schema>'`. Implementation note:
# Claude Code translates `--json-schema` to a synthetic
# `StructuredOutput` tool whose `input_schema` is the user's
# schema, then surfaces the tool_use input as
# `structured_output: {...}` on the trailing `result` event.
# This row tests that proxy-side handling of that tool round-
# trips end-to-end. It does NOT test Anthropic's server-side
# `output_config.schema` parameter (a separate feature used
# internally by Claude Code for session-title generation) --
# `output_config` regressions surface in the HTTP-probe rows.
- id: count_tokens
name: count_tokens endpoint
# HTTP-probe row. Sends a direct POST to
# `{proxy}/v1/messages/count_tokens` for each Claude tier and
# asserts the response is shaped `{"input_tokens": <positive
# int>}`. The CLI uses this endpoint internally but never
# surfaces its result in stream-json, so the only way to test
# the proxy's handling of it is to hit it directly. LiteLLM has
# shipped fixes here (e.g. Claude Code release-notes 2.1.121
# "Vertex AI count_tokens returning 400 errors for proxy
# gateways"), which is exactly the regression class this row
# is meant to catch.
- id: tool_search
name: Tool search (MCP discovery)
# HTTP-probe row. Sends a request whose `tools` array includes
# a `tool_search_tool_regex_20251119` discovery tool and asserts
# the proxy + upstream accept it. This verifies LiteLLM's
# per-provider beta-header translation
# (`advanced-tool-use-2025-11-20` for Anthropic/Azure,
# `tool-search-tool-2025-10-19` for Vertex/Bedrock) is wired up.
# We deliberately don't try to trigger Claude Code's MCP-fan-out
# heuristic via `--mcp-config` -- that would couple the row to
# an internal behavior threshold that changes between Claude
# Code releases. The HTTP probe hits the bug surface LiteLLM
# has actually shipped fixes for (2.1.117, 2.1.72, 2.1.70 per
# the Claude Code release notes).
- id: long_context_1m
name: Long context (1M)
# Sends a ~210k-token padded prompt with the
# `context-1m-2025-08-07` beta header. Just-above the standard
# 200k context window so the request can only succeed when the
# beta header makes it all the way through the proxy to the
# upstream. Haiku 4.5 is intentionally omitted from this row's
# model list (its window is 200k); Sonnet 4.6 and Opus 4.7 are
# the only tiers exercised. Costs roughly $4/cell/run --
# tighten the prompt-token target if pricing changes meaningfully.

View file

@ -0,0 +1,198 @@
"""Matrix JSON Builder.
Pure-function module that consumes the pytest-produced `compat-results.json`,
the manifest, and run metadata, and emits the final `compatibility-matrix.json`
conforming to the schema published in the PRD.
This module is deliberately free of subprocess, network, or filesystem side
effects in its public API the public entry points take pre-loaded inputs
and return data structures, so they can be exercised by golden-file tests
without I/O. A small `build_from_paths()` convenience wrapper does the I/O
for callers that need it (the daily-cron publisher).
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Dict, List, Mapping, Optional, Sequence
import yaml
SCHEMA_VERSION = "1"
VALID_STATUSES = {"pass", "fail", "not_applicable", "not_tested"}
class ManifestError(ValueError):
"""Raised when `manifest.yaml` is malformed."""
class ResultsError(ValueError):
"""Raised when the pytest results artifact is malformed."""
def load_manifest(path: Path) -> Dict[str, Any]:
"""Load and validate `manifest.yaml`.
Returns a dict with keys: schema_version, providers, features. Raises
ManifestError on missing fields or schema mismatch.
"""
raw = yaml.safe_load(path.read_text())
if not isinstance(raw, dict):
raise ManifestError(f"manifest at {path} is not a mapping")
schema_version = str(raw.get("schema_version", ""))
if schema_version != SCHEMA_VERSION:
raise ManifestError(
f"manifest schema_version {schema_version!r} does not match "
f"builder version {SCHEMA_VERSION!r}"
)
providers = raw.get("providers")
if not isinstance(providers, list) or not providers:
raise ManifestError("manifest.providers must be a non-empty list")
features = raw.get("features")
if not isinstance(features, list) or not features:
raise ManifestError("manifest.features must be a non-empty list")
for feature in features:
if not isinstance(feature, dict):
raise ManifestError("each feature must be a mapping")
if not feature.get("id") or not feature.get("name"):
raise ManifestError("each feature must have id and name")
return raw
def load_results(path: Path) -> List[Dict[str, Any]]:
"""Load the pytest results artifact and return its `results` list."""
raw = json.loads(path.read_text())
if not isinstance(raw, dict) or not isinstance(raw.get("results"), list):
raise ResultsError(f"results artifact at {path} has no `results` list")
return raw["results"]
def build_matrix(
*,
manifest: Mapping[str, Any],
results: Sequence[Mapping[str, Any]],
litellm_version: str,
claude_code_version: str,
generated_at: str,
) -> Dict[str, Any]:
"""Build the published matrix JSON from pre-loaded inputs.
Empty cells (no test ran for a (feature, provider) and no
`not_applicable` was declared) are filled in with `not_tested`. If
multiple results report on the same cell e.g. a per-feature test
file containing one parametrize per Claude model the cell aggregates
to `pass` only if every model passed; otherwise `fail` with the first
breaking model surfaced in the error.
"""
providers: List[str] = list(manifest["providers"])
feature_specs: List[Dict[str, Any]] = list(manifest["features"])
grouped: Dict[tuple, List[Dict[str, Any]]] = {}
for entry in results:
if not isinstance(entry, Mapping):
continue
feature_id = entry.get("feature_id")
provider = entry.get("provider")
result = entry.get("result")
if not feature_id or not provider or not isinstance(result, Mapping):
continue
if result.get("status") not in VALID_STATUSES:
continue
grouped.setdefault((feature_id, provider), []).append(dict(result))
features_out: List[Dict[str, Any]] = []
for spec in feature_specs:
feature_id = spec["id"]
cells: Dict[str, Dict[str, Any]] = {}
for provider in providers:
cell_results = grouped.get((feature_id, provider), [])
cells[provider] = _aggregate_cell(cell_results)
features_out.append(
{
"id": feature_id,
"name": spec["name"],
"providers": cells,
}
)
return {
"schema_version": SCHEMA_VERSION,
"generated_at": generated_at,
"litellm_version": litellm_version,
"claude_code_version": claude_code_version,
"providers": providers,
"features": features_out,
}
def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
"""Aggregate a list of per-model results into a single cell status.
Order of precedence (most informative wins):
- Any `fail` cell is `fail` with every failing model's error
joined by `"; "` so a multi-tier breakage doesn't silently hide
all but the first error from the published matrix.
- Any `pass` cell is `pass`. A mix of (pass, not_applicable)
e.g. a tier where the feature isn't supported alongside tiers
where it works surfaces as `pass` so the published cell
reflects that the feature *does* work on this provider rather
than silently demoting it to `not_applicable` and discarding
the passing tiers.
- All `not_applicable` cell is `not_applicable` with the first
row's reason.
- empty / nothing recognized `not_tested`.
`not_tested` rows are treated as absent data: they're dropped before
aggregation so a mix of (pass, not_tested) e.g. from a partial
crash or a test that explicitly recorded "this tier didn't run"
still surfaces the passing tiers rather than silently demoting the
whole cell to `not_tested`. A cell is only `not_tested` when *every*
row is `not_tested` (or there are no rows at all).
"""
if not results:
return {"status": "not_tested"}
observed = [r for r in results if r.get("status") != "not_tested"]
if not observed:
return {"status": "not_tested"}
failures = [r for r in observed if r.get("status") == "fail"]
if failures:
errors = [str(r.get("error", "test failed")) for r in failures]
return {"status": "fail", "error": "; ".join(errors)}
if any(r.get("status") == "pass" for r in observed):
return {"status": "pass"}
if all(r.get("status") == "not_applicable" for r in observed):
return {
"status": "not_applicable",
"reason": str(observed[0].get("reason", "not applicable")),
}
return {"status": "not_tested"}
def build_from_paths(
*,
manifest_path: Path,
results_path: Path,
litellm_version: str,
claude_code_version: str,
generated_at: str,
output_path: Optional[Path] = None,
) -> Dict[str, Any]:
"""I/O wrapper around build_matrix used by the publisher script."""
manifest = load_manifest(manifest_path)
results = load_results(results_path)
matrix = build_matrix(
manifest=manifest,
results=results,
litellm_version=litellm_version,
claude_code_version=claude_code_version,
generated_at=generated_at,
)
if output_path is not None:
output_path.write_text(json.dumps(matrix, indent=2, sort_keys=False) + "\n")
return matrix

View file

@ -0,0 +1,176 @@
"""pdf_input x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, write a tiny valid PDF to disk, allow the built-in `Read`
tool, and ask Claude to read the PDF and report what it contains.
The Read tool inlines the PDF bytes as `document` content blocks on the
next assistant turn, which is exactly the gateway path we want to
exercise: it's distinct from image content blocks (which are tested in
`vision/`) and uses a different transformation in LiteLLM's Anthropic
provider. We assert the upstream produces a non-empty reply that
references the contents of the PDF proving the proxy preserved the
document content block end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/pdf_input/test_anthropic.py
^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
# Smallest valid PDF that renders a single visible word ("PONG"). Built
# inline rather than checked in as a binary fixture so the test stays
# self-contained and the marker word is easy to grep for in CI logs.
# The structure is a hand-crafted single-page PDF with one Helvetica
# text show; offsets are computed at write time so the xref table
# stays consistent regardless of platform line endings.
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`.
We construct the PDF imperatively because `pypdf`/`reportlab` are
not in the test deps and we want the cell to work in a clean
environment. The xref offsets are recomputed for each `marker`
length so the file stays well-formed.
"""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
# Page content stream: position the text and show the marker.
# `BT ... ET` is a text object; `Tf` selects font, `Td` moves
# the cursor, `Tj` paints a string.
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
# Fix up the /Length on the content stream to match its body.
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_anthropic(compat_result, tmp_path):
"""Drive the `claude` CLI against the LiteLLM proxy with a PDF
attached via the Read tool and assert the reply references it."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
# The strongest gateway-level signal we can assert without
# parsing every event type: the model's final user-visible
# reply names the marker word that only the PDF carries.
# If the proxy dropped the `document` content block, the
# model has no way to produce this token.
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,146 @@
"""pdf_input x Microsoft Foundry (Azure).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
write a tiny valid PDF to disk, allow the built-in `Read` tool, and
ask Claude to read the PDF and report what it contains.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/pdf_input/test_azure.py
^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_azure(compat_result, tmp_path):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,152 @@
"""pdf_input x Bedrock (Converse).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the `Converse` API, write a tiny
valid PDF to disk, allow the built-in `Read` tool, and ask Claude to
read the PDF and report what it contains.
Bedrock Converse expresses documents via its own
`document = { format, name, source: { bytes } }` shape; this cell
catches gateway regressions where the proxy fails to translate
Anthropic's `document` content block to Converse's document format
(or vice versa on the response).
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/pdf_input/test_bedrock_converse.py
^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_bedrock_converse(compat_result, tmp_path):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,151 @@
"""pdf_input x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
write a tiny valid PDF to disk, allow the built-in `Read` tool, and
ask Claude to read the PDF and report what it contains.
Bedrock InvokeModel for Anthropic models accepts the native
`document` content block shape; this cell catches gateway regressions
where the proxy drops or mis-encodes the document content block on
the way through (e.g. base64-only encoding, missing media_type, etc.).
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/pdf_input/test_bedrock_invoke.py
^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_bedrock_invoke(compat_result, tmp_path):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,146 @@
"""pdf_input x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to GCP Vertex AI, write a tiny valid PDF to disk, allow
the built-in `Read` tool, and ask Claude to read the PDF and report
what it contains.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/pdf_input/test_vertex_ai.py
^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
PDF_MARKER = "PONG"
def _build_minimal_pdf(marker: str) -> bytes:
"""Return a single-page PDF whose only visible text is `marker`."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>"
),
(
b"<< /Length %d >>\nstream\nBT /F1 24 Tf 50 100 Td ("
+ marker.encode("ascii")
+ b") Tj ET\nendstream"
),
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
body_open = objects[3].index(b"\nstream\n") + len(b"\nstream\n")
body_close = objects[3].index(b"\nendstream")
body_len = body_close - body_open
objects[3] = (
b"<< /Length "
+ str(body_len).encode("ascii")
+ b" >>\nstream\n"
+ objects[3][body_open:body_close]
+ b"\nendstream"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for i, obj in enumerate(objects, start=1):
offsets.append(len(out))
out += f"{i} 0 obj\n".encode("ascii") + obj + b"\nendobj\n"
xref_offset = len(out)
out += b"xref\n0 %d\n" % (len(objects) + 1)
out += b"0000000000 65535 f \n"
for off in offsets:
out += f"{off:010d} 00000 n \n".encode("ascii")
out += (
b"trailer\n<< /Size "
+ str(len(objects) + 1).encode("ascii")
+ b" /Root 1 0 R >>\nstartxref\n"
+ str(xref_offset).encode("ascii")
+ b"\n%%EOF\n"
)
return bytes(out)
def test_pdf_input_vertex_ai(compat_result, tmp_path):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
pdf_path = tmp_path / "marker.pdf"
pdf_path.write_bytes(_build_minimal_pdf(PDF_MARKER))
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt=(
f"Use the Read tool to read the file at {pdf_path}. "
"Report the single word that appears in the document."
),
base_url=base_url,
api_key=api_key,
extra_args=["--allowed-tools", "Read"],
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if PDF_MARKER not in outcome.text.upper():
error = (
f"[{model}] reply did not reference the PDF marker {PDF_MARKER!r}; "
f"got: {outcome.text.strip()!r}"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,150 @@
"""Claude Code PR-Gate Version Resolver.
Resolves the `@anthropic-ai/claude-code` npm version that the PR-gate CI
job installs. Selects the newest version (by publish timestamp) whose
publish timestamp is at least 3 days old. The 3-day window is a security
review buffer see PRD #26476, "Version resolvers".
Two surfaces:
- ``resolve_pr_gate_version(...)`` the importable function. Accepts
pre-fetched npm metadata (for unit tests) or a custom ``fetcher``
callable. The default fetcher hits the public npm registry.
- ``python -m claude_code.pr_gate_version_resolver`` prints the
resolved version string to stdout, suitable for piping into a shell
``$(...)`` substitution inside the CircleCI job.
The CLI form is what CircleCI runs at job start; engineers reading the
job log can see the selected version on a single line above the
``npm install -g`` step (acceptance criterion: "the selected Claude
Code version is logged in the CI output").
"""
from __future__ import annotations
import json
import sys
import urllib.request
from datetime import datetime, timedelta, timezone
from typing import Callable, Mapping, Optional
PACKAGE_NAME = "@anthropic-ai/claude-code"
NPM_REGISTRY_URL = "https://registry.npmjs.org/{package}"
DEFAULT_MIN_AGE = timedelta(days=3)
DEFAULT_FETCH_TIMEOUT_SECONDS = 30
# npm's `time` map mixes per-version timestamps with these meta keys.
_TIME_META_KEYS = frozenset({"created", "modified"})
class NoEligibleVersionError(RuntimeError):
"""Raised when no version in the npm metadata satisfies the min-age cutoff."""
def _parse_npm_timestamp(value: str) -> datetime:
"""Parse the ISO-8601 timestamps npm emits (always UTC, may use ``Z``)."""
if value.endswith("Z"):
value = value[:-1] + "+00:00"
parsed = datetime.fromisoformat(value)
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=timezone.utc)
return parsed
def _default_fetcher(package_name: str) -> dict:
"""Fetch the npm packument for ``package_name`` over HTTPS.
Uses urllib (stdlib) so this module has no extra dependencies in the
CI environment. Returns the raw JSON dict.
"""
# urllib.parse.quote would encode the leading '@' / '/' which the
# npm registry expects literally; do a minimal hand-roll instead.
url = NPM_REGISTRY_URL.format(package=package_name.replace("/", "%2F"))
req = urllib.request.Request(url, headers={"Accept": "application/json"})
with urllib.request.urlopen( # noqa: S310 — registry URL is constant
req, timeout=DEFAULT_FETCH_TIMEOUT_SECONDS
) as response:
body = response.read().decode("utf-8")
return json.loads(body)
def resolve_pr_gate_version(
*,
metadata: Optional[Mapping] = None,
fetcher: Optional[Callable[[str], Mapping]] = None,
as_of: Optional[datetime] = None,
min_age: timedelta = DEFAULT_MIN_AGE,
package_name: str = PACKAGE_NAME,
) -> str:
"""Return the newest npm version of ``package_name`` published >= ``min_age`` ago.
"Newest" means newest by **publish time**, not semver string order
if a patch lands on an older major after a newer release, the
patched line is the eligible one.
Args:
metadata: Pre-fetched npm packument (skips the HTTP call). Useful
for unit tests.
fetcher: Callable taking a package name and returning the
packument. Defaults to a stdlib HTTPS fetcher.
as_of: The clock used to decide whether a version is "old
enough". Defaults to ``datetime.now(timezone.utc)``.
min_age: Minimum publish age. Defaults to 3 days.
package_name: Defaults to ``@anthropic-ai/claude-code``.
Raises:
NoEligibleVersionError: when no version in the registry meets
the age cutoff.
"""
if metadata is None:
fetch = fetcher or _default_fetcher
metadata = fetch(package_name)
times = metadata.get("time") or {}
if as_of is None:
as_of = datetime.now(timezone.utc)
cutoff = as_of - min_age
eligible: list[tuple[datetime, str]] = []
for version, raw_ts in times.items():
if version in _TIME_META_KEYS:
continue
if not isinstance(raw_ts, str):
continue
if "-" in version:
continue
published = _parse_npm_timestamp(raw_ts)
if published <= cutoff:
eligible.append((published, version))
if not eligible:
raise NoEligibleVersionError(
f"no version of {package_name} is at least {min_age} old "
f"as of {as_of.isoformat()}"
)
eligible.sort(key=lambda pair: pair[0], reverse=True)
return eligible[0][1]
def _main(argv: list[str]) -> int:
"""Print the resolved version to stdout. Exit code 0 on success.
Stderr carries the human-readable announcement so the version can be
captured cleanly with ``$(python -m ...)`` in shell.
"""
try:
version = resolve_pr_gate_version()
except Exception as exc: # noqa: BLE001 — CLI surface, want everything
print(f"pr_gate_version_resolver: {exc}", file=sys.stderr) # noqa: T201
return 1
print( # noqa: T201
f"pr_gate_version_resolver: selected {PACKAGE_NAME}@{version}",
file=sys.stderr,
)
print(version) # noqa: T201
return 0
if __name__ == "__main__":
raise SystemExit(_main(sys.argv[1:]))

View file

@ -0,0 +1,124 @@
"""prompt_caching_1h x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, opt into the 1-hour cache TTL via `ENABLE_PROMPT_CACHING_1H`,
and assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0 i.e.
the proxy preserved Claude Code's `cache_control: { ttl: "1h" }`
annotations end-to-end and the upstream actually honored them.
This is the 1-hour-TTL companion to `prompt_caching_5m/`. It exists as
its own cell because the 1h TTL travels through the proxy with a
distinct `cache_control` shape (and a distinct beta-header gate on
some providers); a regression that strips or downgrades the TTL on the
way through is invisible to the 5m cell, which would still see cache
hits with a default-TTL annotation.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_1h/test_anthropic.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
# Per the changelog (2.1.108): `ENABLE_PROMPT_CACHING_1H` flips Claude
# Code from the default 5-minute cache TTL to a 1-hour TTL on the
# `cache_control` annotations it adds to the system prompt and the
# most recent user turn. Setting it here is what we are validating
# the proxy faithfully forwards to the upstream.
CACHE_1H_ENV = {"ENABLE_PROMPT_CACHING_1H": "1"}
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
"""Return cache_creation_input_tokens + cache_read_input_tokens from
the upstream usage block, or 0 if the keys are missing."""
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_1h_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with the 1h
TTL opt-in env var set, and assert the upstream usage block
surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
extra_env=CACHE_1H_ENV,
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens with "
"ENABLE_PROMPT_CACHING_1H=1; the proxy likely stripped the "
"1h TTL beta header or rejected the cache_control shape"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,104 @@
"""prompt_caching_1h x Microsoft Foundry (Azure).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
opt into the 1-hour cache TTL via `ENABLE_PROMPT_CACHING_1H`, and
assert the upstream's usage block reports a non-zero cache token count.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_1h/test_azure.py
^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
CACHE_1H_ENV = {"ENABLE_PROMPT_CACHING_1H": "1"}
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_1h_azure(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
extra_env=CACHE_1H_ENV,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens with "
"ENABLE_PROMPT_CACHING_1H=1"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,112 @@
"""prompt_caching_1h x Bedrock (Converse).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the `Converse` API, opt into the
1-hour cache TTL via `ENABLE_PROMPT_CACHING_1H`, and assert the
upstream's usage block reports a non-zero cache token count.
Bedrock Converse expresses prompt caching via `cachePoint` markers in
the message list, with TTL controlled out-of-band; this cell catches
proxy regressions where the 1h opt-in fails to translate into the
correct Converse cache configuration.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_1h/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
CACHE_1H_ENV = {
"ENABLE_PROMPT_CACHING_1H": "1",
"ENABLE_PROMPT_CACHING_1H_BEDROCK": "1",
}
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_1h_bedrock_converse(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
extra_env=CACHE_1H_ENV,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens with "
"1h-TTL opt-in env vars set"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,117 @@
"""prompt_caching_1h x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
opt into the 1-hour cache TTL via `ENABLE_PROMPT_CACHING_1H`, and
assert the upstream's usage block reports a non-zero cache token count.
Bedrock historically gated 1h prompt caching behind a separate
`ENABLE_PROMPT_CACHING_1H_BEDROCK` env var (see 2.1.108: deprecated but
still honored). The proxy must accept either env var and forward an
appropriate `cache_control` shape to the Bedrock InvokeModel endpoint;
this cell catches regressions where the TTL is silently downgraded to
5 minutes on the way through.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_1h/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
# Set both the modern and the deprecated-but-honored Bedrock var so we
# match whichever code path the proxy is following.
CACHE_1H_ENV = {
"ENABLE_PROMPT_CACHING_1H": "1",
"ENABLE_PROMPT_CACHING_1H_BEDROCK": "1",
}
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_1h_bedrock_invoke(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
extra_env=CACHE_1H_ENV,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens with "
"1h-TTL opt-in env vars set; the proxy likely stripped or "
"downgraded the cache_control TTL"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,104 @@
"""prompt_caching_1h x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to GCP Vertex AI, opt into the 1-hour cache TTL via
`ENABLE_PROMPT_CACHING_1H`, and assert the upstream's usage block
reports a non-zero cache token count.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_1h/test_vertex_ai.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
CACHE_1H_ENV = {"ENABLE_PROMPT_CACHING_1H": "1"}
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_1h_vertex_ai(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
extra_env=CACHE_1H_ENV,
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens with "
"ENABLE_PROMPT_CACHING_1H=1"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,113 @@
"""prompt_caching_5m x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, and assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0 i.e.
the proxy preserves Claude Code's `cache_control` annotations end-to-end
and the upstream actually honored them. This is the 5-minute (default)
cache TTL row.
Claude Code itself sets `cache_control: { type: "ephemeral" }` on the
system prompt and the most recent user turn for every request, so a
single live invocation is enough to surface a cache-creation count on
the first call and a cache-read count on a warm follow-up call.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_5m/test_anthropic.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
"""Return cache_creation_input_tokens + cache_read_input_tokens from
the upstream usage block, or 0 if the keys are missing."""
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_5m_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream usage block surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens; "
"expected cache_control on the system prompt to produce a non-zero "
"cache_creation_input_tokens or cache_read_input_tokens"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,111 @@
"""prompt_caching_5m x Azure (Microsoft Foundry).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models hosted in Microsoft Foundry on
Azure, and assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0.
Foundry's Anthropic deployments honor the default 5-minute ephemeral
`cache_control` exactly like anthropic.com. The 1-hour `scope: "global"`
variant is *not* supported on Foundry LiteLLM strips that field
before forwarding (see `_remove_scope_from_cache_control` in
`litellm/llms/azure_ai/anthropic/messages_transformation.py`) but
this row exercises the 5-minute TTL only, so that quirk does not apply.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_5m/test_azure.py
^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_5m_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream usage block surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens; "
"expected cache_control on the system prompt to produce a non-zero "
"cache_creation_input_tokens or cache_read_input_tokens"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,104 @@
"""prompt_caching_5m x Bedrock (Converse).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the unified `Converse` API path, and
assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_5m/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_5m_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream usage block surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens; "
"expected cache_control on the system prompt to produce a non-zero "
"cache_creation_input_tokens or cache_read_input_tokens"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,104 @@
"""prompt_caching_5m x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
and assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_5m/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_5m_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream usage block surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens; "
"expected cache_control on the system prompt to produce a non-zero "
"cache_creation_input_tokens or cache_read_input_tokens"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,104 @@
"""prompt_caching_5m x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models on Google Cloud Vertex AI, and
assert that the upstream's usage block reports either
`cache_creation_input_tokens` or `cache_read_input_tokens` > 0.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/prompt_caching_5m/test_vertex_ai.py
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Optional
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
def _cache_tokens(usage: Optional[Mapping[str, Any]]) -> int:
if not isinstance(usage, Mapping):
return 0
creation = usage.get("cache_creation_input_tokens") or 0
read = usage.get("cache_read_input_tokens") or 0
try:
return int(creation) + int(read)
except (TypeError, ValueError):
return 0
def test_prompt_caching_5m_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream usage block surfaces a non-zero cache token count."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt="Reply with the single word 'pong' and nothing else.",
base_url=base_url,
api_key=api_key,
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if _cache_tokens(outcome.usage) <= 0:
error = (
f"[{model}] usage block reported zero cache tokens; "
"expected cache_control on the system prompt to produce a non-zero "
"cache_creation_input_tokens or cache_read_input_tokens"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,358 @@
"""Cross-process token-bucket rate limiter for the Claude Code compat suite.
The compat matrix runs 75 live `claude` CLI invocations (25 cells × 3
Claude tiers per cell). When pytest-xdist fans these out across worker
processes, each worker would maintain its own in-memory rate limiter
and the *aggregate* request rate hitting any one upstream provider
would be `workers × per-worker rate` exactly the situation that
trips Anthropic / Azure / Bedrock / Vertex 429s in the middle of a
matrix run and silently flips green cells red.
The fix is a **shared, cross-process** token bucket per provider,
backed by a small JSON state file and an OS-level `flock`. Each
`run_claude` invocation acquires one token (sleeping if the bucket is
empty) before launching the CLI; refills happen lazily based on wall
time, so workers can be killed and restarted without losing or
double-spending budget.
Configuration is driven entirely by environment variables so a
binary-search workflow can shift per-provider rates without code
edits:
LITELLM_COMPAT_RATE_ANTHROPIC (req/s, default 5.0)
LITELLM_COMPAT_RATE_AZURE (req/s, default 5.0)
LITELLM_COMPAT_RATE_VERTEX_AI (req/s, default 5.0)
LITELLM_COMPAT_RATE_BEDROCK_CONVERSE (req/s, default 5.0)
LITELLM_COMPAT_RATE_BEDROCK_INVOKE (req/s, default 5.0)
LITELLM_COMPAT_RATE_BURST (per-bucket burst override;
default = rate)
LITELLM_COMPAT_RATE_STATE_DIR (state file directory;
default = $TMPDIR/litellm-claude-compat-ratelimit)
A rate of 0 (or any non-positive value) disables throttling for that
provider useful when you trust the upstream to handle the burst or
when running unit-test-shaped workloads that never actually hit the
network.
The provider id is inferred from the model id by `infer_provider`,
mirroring the matrix's column layout (`anthropic`, `azure`,
`vertex_ai`, `bedrock_converse`, `bedrock_invoke`).
"""
from __future__ import annotations
import contextlib
import json
import math
import os
import tempfile
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, Iterator, Mapping, Optional
# `fcntl` is POSIX-only; the suite is Linux/macOS only, so we don't
# attempt a Windows fallback. Importing at module load fails fast on
# the (currently non-existent) Windows runner so we don't silently
# degrade to no-locking behavior.
import fcntl
PROVIDER_ANTHROPIC = "anthropic"
PROVIDER_AZURE = "azure"
PROVIDER_VERTEX_AI = "vertex_ai"
PROVIDER_BEDROCK_CONVERSE = "bedrock_converse"
PROVIDER_BEDROCK_INVOKE = "bedrock_invoke"
ALL_PROVIDERS = (
PROVIDER_ANTHROPIC,
PROVIDER_AZURE,
PROVIDER_VERTEX_AI,
PROVIDER_BEDROCK_CONVERSE,
PROVIDER_BEDROCK_INVOKE,
)
DEFAULT_RATE = 5.0 # req/s per provider, conservative starting point
RATE_ENV_PREFIX = "LITELLM_COMPAT_RATE_"
BURST_ENV = "LITELLM_COMPAT_RATE_BURST"
STATE_DIR_ENV = "LITELLM_COMPAT_RATE_STATE_DIR"
DEFAULT_STATE_DIR_NAME = "litellm-claude-compat-ratelimit"
def infer_provider(model: str) -> str:
"""Map a model alias to its compat-matrix provider id.
The matrix column layout is fixed; aliases registered in the proxy
encode the provider via a suffix (`-bedrock-converse`,
`-bedrock-invoke`, `-azure`, `-vertex`) or its absence (Anthropic).
Order matters: the bedrock suffixes both contain `bedrock`, so we
test the more-specific ones first.
"""
if not model:
raise ValueError("model must be a non-empty string")
lower = model.lower()
if lower.endswith("-bedrock-converse"):
return PROVIDER_BEDROCK_CONVERSE
if lower.endswith("-bedrock-invoke"):
return PROVIDER_BEDROCK_INVOKE
if lower.endswith("-azure"):
return PROVIDER_AZURE
if lower.endswith("-vertex"):
return PROVIDER_VERTEX_AI
return PROVIDER_ANTHROPIC
@dataclass(frozen=True)
class ProviderConfig:
"""Static config snapshot for one provider's bucket.
Captured up front (rather than re-read per acquire) so the limiter's
behavior in a single process is stable even if env vars are mutated
mid-run. A fresh `RateLimiter` picks up env changes on construction.
"""
rate_per_sec: float
burst: float
@property
def enabled(self) -> bool:
return self.rate_per_sec > 0 and self.burst > 0
def load_config(
env: Optional[Mapping[str, str]] = None,
) -> Dict[str, ProviderConfig]:
"""Build a {provider: ProviderConfig} from env, applying defaults.
Parsing failures fall back to the default rate rather than
crashing the test session a typo in `LITELLM_COMPAT_RATE_AZURE`
should not silently disable throttling, but it also shouldn't
abort 75 live tests with a `ValueError` ten minutes in.
"""
src = env if env is not None else os.environ
def _as_float(value: Optional[str], default: float) -> float:
if value is None or value == "":
return default
try:
return float(value)
except (TypeError, ValueError):
return default
burst_override = _as_float(src.get(BURST_ENV), -1.0)
out: Dict[str, ProviderConfig] = {}
for provider in ALL_PROVIDERS:
env_key = RATE_ENV_PREFIX + provider.upper()
rate = _as_float(src.get(env_key), DEFAULT_RATE)
burst = burst_override if burst_override > 0 else max(rate, 1.0)
out[provider] = ProviderConfig(rate_per_sec=rate, burst=burst)
return out
def _state_dir(env: Optional[Mapping[str, str]] = None) -> Path:
"""Resolve the directory holding per-provider state files.
A user-supplied `LITELLM_COMPAT_RATE_STATE_DIR` wins for tests and
container environments where `$TMPDIR` may be ephemeral or shared
in surprising ways. The directory is created lazily with
`parents=True, exist_ok=True` so first-run setup needs no
fixture wiring.
"""
src = env if env is not None else os.environ
explicit = src.get(STATE_DIR_ENV)
if explicit:
return Path(explicit)
return Path(tempfile.gettempdir()) / DEFAULT_STATE_DIR_NAME
class RateLimiter:
"""Cross-process token bucket per provider.
Each `acquire(provider)` call:
1. Opens (creating if needed) `<state_dir>/<provider>.json`.
2. Holds an exclusive `flock` while reading + updating the
{tokens, last_refill} state.
3. Refills tokens based on `now - last_refill`, capped at burst.
4. If tokens >= 1, subtracts one and returns immediately.
5. Otherwise, computes the wall-time delay needed to earn a
single token at the configured rate, releases the lock, and
sleeps. After sleeping it retries staying under the lock
while sleeping would serialize all workers behind whichever
one held it longest.
This bounds the *aggregate* req/s seen by the upstream, regardless
of how many xdist workers, threads, or processes are concurrently
running tests against the same provider.
A `_clock` / `_sleep` injection seam keeps the unit tests fast and
deterministic; production callers should never override either.
"""
def __init__(
self,
config: Optional[Mapping[str, ProviderConfig]] = None,
state_dir: Optional[Path] = None,
clock: Optional[callable] = None,
sleep: Optional[callable] = None,
) -> None:
self._config = dict(config) if config is not None else load_config()
self._state_dir = Path(state_dir) if state_dir is not None else _state_dir()
self._clock = clock or time.monotonic
self._sleep = sleep or time.sleep
# `_state_dir.mkdir` once on construction is fine; concurrent
# workers all racing to create the same directory is benign.
self._state_dir.mkdir(parents=True, exist_ok=True)
def acquire(self, provider: str) -> float:
"""Block until one token is available for `provider`.
Returns the cumulative wall-time spent waiting (0.0 when the
bucket had budget and we returned immediately). Callers can
log this to attribute slow cells to throttling vs. upstream
latency same role `DriverResult.duration_ms` plays for the
actual CLI invocation.
"""
cfg = self._config.get(provider)
if cfg is None or not cfg.enabled:
return 0.0
path = self._state_path(provider)
total_waited = 0.0
while True:
now = self._clock()
sleep_for = self._try_consume(path, cfg, now)
if sleep_for <= 0:
return total_waited
self._sleep(sleep_for)
total_waited += sleep_for
def _state_path(self, provider: str) -> Path:
return self._state_dir / f"{provider}.json"
def _try_consume(self, path: Path, cfg: ProviderConfig, now: float) -> float:
"""Atomically refill and try to take one token.
Returns 0.0 if a token was consumed, or a positive sleep
duration (seconds) if the caller must wait before retrying.
We hold an exclusive `flock` only across the read-modify-write
of the JSON state never across a `sleep` so workers don't
serialize while one of them is parked.
"""
# `os.open` + `os.O_CREAT | os.O_RDWR` gives us a fd we can
# both lock and read/write through. Opening with `"a+"` then
# seeking is equivalent but uglier; this version is closer to
# the canonical flock recipe.
fd = os.open(path, os.O_CREAT | os.O_RDWR, 0o600)
try:
fcntl.flock(fd, fcntl.LOCK_EX)
try:
tokens, last_refill = self._read_state(fd, cfg, now)
tokens, last_refill = self._refill(tokens, last_refill, now, cfg)
if tokens >= 1.0:
tokens -= 1.0
self._write_state(fd, tokens, last_refill)
return 0.0
# Not enough budget. Persist the refilled state so a
# subsequent caller doesn't have to redo the math, then
# release the lock and tell the caller how long to
# sleep before retrying.
self._write_state(fd, tokens, last_refill)
deficit = 1.0 - tokens
# `deficit / rate` seconds will earn exactly enough
# for one token. Add a tiny safety margin so we don't
# wake up nanoseconds early and spin.
return (deficit / cfg.rate_per_sec) + 1e-3
finally:
fcntl.flock(fd, fcntl.LOCK_UN)
finally:
os.close(fd)
@staticmethod
def _read_state(fd: int, cfg: ProviderConfig, now: float) -> tuple:
"""Read {tokens, last_refill} from `fd`, defaulting to a full
bucket on a missing/empty/corrupt file.
New files start full so the very first request never waits;
corrupt files are treated like new files because the
alternative refusing to run is worse than briefly
over-spending one bucket's worth of budget.
"""
os.lseek(fd, 0, os.SEEK_SET)
raw = os.read(fd, 4096).decode("utf-8")
if not raw.strip():
return cfg.burst, now
try:
obj = json.loads(raw)
tokens = float(obj.get("tokens", cfg.burst))
last_refill = float(obj.get("last_refill", now))
return tokens, last_refill
except (ValueError, TypeError):
return cfg.burst, now
@staticmethod
def _refill(
tokens: float, last_refill: float, now: float, cfg: ProviderConfig
) -> tuple:
"""Apply elapsed time to the bucket, capped at burst.
Negative elapsed (clock went backward, e.g. across host
sleep/resume or a manually-tweaked monotonic mock) is clamped
to zero so we never *remove* tokens.
"""
elapsed = max(0.0, now - last_refill)
tokens = min(cfg.burst, tokens + elapsed * cfg.rate_per_sec)
return tokens, now
@staticmethod
def _write_state(fd: int, tokens: float, last_refill: float) -> None:
payload = json.dumps({"tokens": tokens, "last_refill": last_refill}).encode(
"utf-8"
)
os.lseek(fd, 0, os.SEEK_SET)
os.ftruncate(fd, 0)
os.write(fd, payload)
# A single process-wide instance is all we need: provider-keyed state
# is stored in files, so multiple `RateLimiter` instances would just
# duplicate the in-process bookkeeping. We expose a getter rather than
# the instance directly so unit tests can install a custom limiter
# scoped to a tmp directory without monkeypatching globals.
_default: Optional[RateLimiter] = None
def get_default_limiter() -> RateLimiter:
global _default
if _default is None:
_default = RateLimiter()
return _default
def reset_default_limiter() -> None:
"""Drop the cached default limiter; the next `get_default_limiter`
call rebuilds it from the current environment.
Useful between unit tests that patch env vars: without this they'd
keep reading the stale config snapshot from the first call.
"""
global _default
_default = None
@contextlib.contextmanager
def use_limiter(limiter: RateLimiter) -> Iterator[RateLimiter]:
"""Temporarily install `limiter` as the process default.
The driver's `run_claude` calls `get_default_limiter()`; tests that
want a controlled tmp-dir-backed limiter use this contextmanager
to swap one in without touching env vars or the on-disk default
state.
"""
global _default
previous = _default
_default = limiter
try:
yield limiter
finally:
_default = previous

View file

@ -0,0 +1,108 @@
#!/usr/bin/env bash
# Run the Claude Code compat matrix end-to-end against a live LiteLLM
# proxy, with per-provider rate limits applied via the cross-process
# token bucket in `tests/e2e/claude_code/rate_limiter.py`.
#
# Designed for binary-searching the ideal X / Y / Z req/s per provider:
# 1. Pick an initial rate (e.g. 5/s for everyone).
# 2. Run this script.
# 3. Read `compat-rate-limit-summary.json` to see whether any provider
# hit a 429-shaped error during the run.
# 4. If a provider has `rate_limited > 0`, halve its rate; else, double it.
# 5. Repeat until the highest no-429 rate is found.
#
# Required env (proxy connection):
# LITELLM_PROXY_BASE_URL e.g. http://localhost:4000
# LITELLM_PROXY_API_KEY e.g. sk-1234
#
# Optional env (rate limits, all default to 5 req/s; 0 disables a column):
# LITELLM_COMPAT_RATE_ANTHROPIC
# LITELLM_COMPAT_RATE_AZURE
# LITELLM_COMPAT_RATE_VERTEX_AI
# LITELLM_COMPAT_RATE_BEDROCK_CONVERSE
# LITELLM_COMPAT_RATE_BEDROCK_INVOKE
# LITELLM_COMPAT_RATE_BURST override per-bucket burst
#
# Optional env (parallelism):
# COMPAT_XDIST_WORKERS passed to `pytest -n` (default: auto)
#
# Optional env (artifacts):
# COMPAT_RESULTS_PATH default: compat-results.json
# COMPAT_RATE_LIMIT_SUMMARY_PATH default: compat-rate-limit-summary.json
set -euo pipefail
if [[ -z "${LITELLM_PROXY_BASE_URL:-}" || -z "${LITELLM_PROXY_API_KEY:-}" ]]; then
echo "error: LITELLM_PROXY_BASE_URL and LITELLM_PROXY_API_KEY must be set" >&2
exit 64
fi
# Reset the cross-process rate-limiter state from any prior run. Stale
# token-bucket files would let a previous run's accumulated budget bleed
# into the new one, which subtly biases the binary search.
state_dir="${LITELLM_COMPAT_RATE_STATE_DIR:-${TMPDIR:-/tmp}/litellm-claude-compat-ratelimit}"
if [[ -d "$state_dir" ]]; then
rm -rf "$state_dir"
fi
# Worker count. `auto` picks one worker per CPU; the rate limiter
# enforces aggregate provider rates regardless of worker count, so
# this is a "go as fast as the limiter allows" knob, not a tuning knob.
workers="${COMPAT_XDIST_WORKERS:-auto}"
# Where the artifacts land. We resolve them now so the summary file is
# always at a known path the caller can grep, even if they didn't set
# the env explicitly.
results_path="${COMPAT_RESULTS_PATH:-compat-results.json}"
summary_path="${COMPAT_RATE_LIMIT_SUMMARY_PATH:-compat-rate-limit-summary.json}"
echo "[run_compat] rates:"
for provider in ANTHROPIC AZURE VERTEX_AI BEDROCK_CONVERSE BEDROCK_INVOKE; do
var="LITELLM_COMPAT_RATE_${provider}"
echo " ${provider}=${!var:-default(5/s)}"
done
echo " BURST=${LITELLM_COMPAT_RATE_BURST:-default(=rate)}"
echo "[run_compat] xdist workers: ${workers}"
echo "[run_compat] results: ${results_path}"
echo "[run_compat] summary: ${summary_path}"
# Run only the per-feature live tests; skip the unit-test directories
# (they're under directories starting with `_`). The dist=loadfile
# scheduler keeps each test file pinned to a single worker, which is
# what we want — every test in a file shares a single ThreadPoolExecutor
# fanout, and we don't gain anything by splitting it across workers.
start=$(date +%s)
set +e
COMPAT_RESULTS_PATH="${results_path}" \
COMPAT_RATE_LIMIT_SUMMARY_PATH="${summary_path}" \
PATH="$HOME/.local/bin:$PATH" \
uv run pytest \
tests/e2e/claude_code/basic_messaging_non_streaming \
tests/e2e/claude_code/basic_messaging_streaming \
tests/e2e/claude_code/thinking \
tests/e2e/claude_code/tool_use \
tests/e2e/claude_code/vision \
tests/e2e/claude_code/prompt_caching_5m \
-n "${workers}" \
--dist=loadfile \
-q \
"$@"
exit_code=$?
set -e
end=$(date +%s)
echo "[run_compat] wall time: $((end - start))s"
# Surface the rate-limit summary inline so a human reader doesn't have
# to `cat` the JSON file. The full file is still on disk for the binary
# search loop.
if [[ -f "${summary_path}" ]]; then
echo "[run_compat] summary: ${summary_path}"
if command -v jq >/dev/null 2>&1; then
jq '.totals, .per_provider' "${summary_path}"
else
cat "${summary_path}"
fi
fi
exit "${exit_code}"

View file

@ -0,0 +1,141 @@
{
"schema_version": "1",
"generated_at": "2026-04-25T00:00:00Z",
"litellm_version": "v1.83.0-stable",
"claude_code_version": "2.1.120",
"providers": [
"anthropic",
"bedrock_invoke",
"bedrock_converse",
"vertex_ai",
"azure"
],
"features": [
{
"id": "basic_messaging_non_streaming",
"name": "Basic messaging (non-streaming)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
},
{
"id": "basic_messaging_streaming",
"name": "Basic messaging (streaming)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
},
{
"id": "tool_use",
"name": "Tool use",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
},
{
"id": "prompt_caching_5m",
"name": "Prompt caching (5m TTL)",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
},
{
"id": "vision",
"name": "Vision",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
},
{
"id": "thinking",
"name": "Thinking",
"providers": {
"anthropic": {
"status": "pass"
},
"bedrock_invoke": {
"status": "pass"
},
"bedrock_converse": {
"status": "pass"
},
"vertex_ai": {
"status": "pass"
},
"azure": {
"status": "pass"
}
}
}
]
}

View file

@ -0,0 +1,220 @@
"""structured_outputs x Anthropic.
Drive the real `claude` CLI in headless mode with the `--json-schema`
flag, route through a LiteLLM proxy aimed at Anthropic, and assert that
the final stream-json `result` event surfaces a `structured_output`
object whose shape matches the schema.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/structured_outputs/test_anthropic.py
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
What this row actually exercises (and what it does not):
`--json-schema` is implemented client-side by Claude Code: the CLI
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
equals the user-supplied JSON Schema, forces the model toward it, and
finally extracts the tool_use input on the trailing `result` event as
`structured_output: {...}`. The proxy never sees `output_config.schema`
in this flow -- it sees a normal `tools` array with one synthetic
tool.
This makes the row a tool-use feature test in disguise. It's still a
distinct row from `tool_use` because:
- The synthetic tool is generated per request from a user schema, not
a developer-declared one. Provider-side bugs that special-case
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
blocks back to lowercase, or stripping the StructuredOutput-only
`additionalProperties: false`) only surface here.
- The success signal lives on the *final* `result` event, not the
intermediate `assistant` events the `tool_use` row checks. A proxy
that drops trailing events (seen in early Bedrock Converse SSE
plumbing) breaks this cell while leaving `tool_use` green.
It is NOT a test of Anthropic's server-side `output_config.schema`
parameter -- that's a different feature used internally by Claude Code
for session-title generation and is not reachable from any CLI flag.
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
the `count_tokens` and other HTTP-probe rows, not here.
Three Claude tiers run in parallel; one `compat_result.add(...)` per
tier so the matrix's "all three must pass" rule applies.
"""
from __future__ import annotations
import json
import os
import re
from typing import Any, Mapping, Optional, Sequence, Tuple
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
# Minimal schema with one required integer field. Kept intentionally
# small -- the matrix tests the *plumbing*, not the model's ability to
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
# integer schema gives every tier (including Haiku) enough headroom
# that schema satisfaction is essentially deterministic, isolating
# failures to the proxy / transport.
SCHEMA = {
"type": "object",
"properties": {"answer": {"type": "integer"}},
"required": ["answer"],
"additionalProperties": False,
}
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
# A prompt the model has no reason to misanswer; we don't check the
# value, but a wrong answer would suggest the structured-output
# pathway is silently degrading reasoning, which is itself worth
# noticing.
PROMPT = "What is 2 + 2? Reply only via the structured output."
def _extract_structured_output(
events: Sequence[Mapping[str, Any]],
) -> Optional[Mapping[str, Any]]:
"""Return the `structured_output` payload from the last `result` event.
Claude Code emits its terminal stream-json line as
`{"type":"result","structured_output":{...},...}` when a request
used `--json-schema` and the model actually produced a valid tool
call. If the model bailed or the proxy ate the trailing events,
`structured_output` is missing -- which is exactly the failure
mode we want this row to surface, so the caller treats `None` as
"feature did not work end-to-end".
"""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
so = event.get("structured_output")
if isinstance(so, Mapping):
return so
return None
def _validate_against_schema(
payload: Mapping[str, Any], schema: Mapping[str, Any]
) -> Optional[str]:
"""Tiny shape validator covering the subset we actually need.
We deliberately do not pull in `jsonschema` as a test dep: the
matrix's success signal is "does the proxy let the synthetic
StructuredOutput tool round-trip end-to-end", and that's
answerable with a presence + type check over `required` keys.
Any malformed schema beyond that would be a Claude Code bug,
not a LiteLLM-proxy bug, so a deeper check would only add false
failures on the wrong axis.
"""
type_map = {
"integer": int,
"number": (int, float),
"string": str,
"boolean": bool,
"array": list,
"object": Mapping,
}
required = schema.get("required") or []
properties = schema.get("properties") or {}
for key in required:
if key not in payload:
return f"missing required key {key!r}"
expected = (properties.get(key) or {}).get("type")
if expected and expected in type_map:
if not isinstance(payload[key], type_map[expected]):
return (
f"key {key!r} has wrong type: "
f"expected {expected}, got {type(payload[key]).__name__}"
)
# bool is a subclass of int in Python; reject `True`/`False`
# when the schema asked for an integer/number.
if expected in ("integer", "number") and isinstance(payload[key], bool):
return f"key {key!r} is a bool but schema asked for {expected}"
return None
def test_structured_outputs_anthropic(compat_result):
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
assert the trailing `result` event contains a schema-conforming
`structured_output`."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt=PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=["--json-schema", SCHEMA_JSON],
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
payload = _extract_structured_output(outcome.events)
if payload is None:
error = (
f"[{model}] no `structured_output` in trailing result event; "
"Claude Code's StructuredOutput tool round-trip did not "
"complete end-to-end through the proxy"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
shape_error = _validate_against_schema(payload, SCHEMA)
if shape_error is not None:
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,220 @@
"""structured_outputs x Azure (Microsoft Foundry).
Drive the real `claude` CLI in headless mode with the `--json-schema`
flag, route through a LiteLLM proxy aimed at Azure (Microsoft Foundry), and assert that
the final stream-json `result` event surfaces a `structured_output`
object whose shape matches the schema.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/structured_outputs/test_azure.py
^^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
What this row actually exercises (and what it does not):
`--json-schema` is implemented client-side by Claude Code: the CLI
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
equals the user-supplied JSON Schema, forces the model toward it, and
finally extracts the tool_use input on the trailing `result` event as
`structured_output: {...}`. The proxy never sees `output_config.schema`
in this flow -- it sees a normal `tools` array with one synthetic
tool.
This makes the row a tool-use feature test in disguise. It's still a
distinct row from `tool_use` because:
- The synthetic tool is generated per request from a user schema, not
a developer-declared one. Provider-side bugs that special-case
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
blocks back to lowercase, or stripping the StructuredOutput-only
`additionalProperties: false`) only surface here.
- The success signal lives on the *final* `result` event, not the
intermediate `assistant` events the `tool_use` row checks. A proxy
that drops trailing events (seen in early Bedrock Converse SSE
plumbing) breaks this cell while leaving `tool_use` green.
It is NOT a test of Anthropic's server-side `output_config.schema`
parameter -- that's a different feature used internally by Claude Code
for session-title generation and is not reachable from any CLI flag.
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
the `count_tokens` and other HTTP-probe rows, not here.
Three Claude tiers run in parallel; one `compat_result.add(...)` per
tier so the matrix's "all three must pass" rule applies.
"""
from __future__ import annotations
import json
import os
import re
from typing import Any, Mapping, Optional, Sequence, Tuple
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
# Minimal schema with one required integer field. Kept intentionally
# small -- the matrix tests the *plumbing*, not the model's ability to
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
# integer schema gives every tier (including Haiku) enough headroom
# that schema satisfaction is essentially deterministic, isolating
# failures to the proxy / transport.
SCHEMA = {
"type": "object",
"properties": {"answer": {"type": "integer"}},
"required": ["answer"],
"additionalProperties": False,
}
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
# A prompt the model has no reason to misanswer; we don't check the
# value, but a wrong answer would suggest the structured-output
# pathway is silently degrading reasoning, which is itself worth
# noticing.
PROMPT = "What is 2 + 2? Reply only via the structured output."
def _extract_structured_output(
events: Sequence[Mapping[str, Any]],
) -> Optional[Mapping[str, Any]]:
"""Return the `structured_output` payload from the last `result` event.
Claude Code emits its terminal stream-json line as
`{"type":"result","structured_output":{...},...}` when a request
used `--json-schema` and the model actually produced a valid tool
call. If the model bailed or the proxy ate the trailing events,
`structured_output` is missing -- which is exactly the failure
mode we want this row to surface, so the caller treats `None` as
"feature did not work end-to-end".
"""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
so = event.get("structured_output")
if isinstance(so, Mapping):
return so
return None
def _validate_against_schema(
payload: Mapping[str, Any], schema: Mapping[str, Any]
) -> Optional[str]:
"""Tiny shape validator covering the subset we actually need.
We deliberately do not pull in `jsonschema` as a test dep: the
matrix's success signal is "does the proxy let the synthetic
StructuredOutput tool round-trip end-to-end", and that's
answerable with a presence + type check over `required` keys.
Any malformed schema beyond that would be a Claude Code bug,
not a LiteLLM-proxy bug, so a deeper check would only add false
failures on the wrong axis.
"""
type_map = {
"integer": int,
"number": (int, float),
"string": str,
"boolean": bool,
"array": list,
"object": Mapping,
}
required = schema.get("required") or []
properties = schema.get("properties") or {}
for key in required:
if key not in payload:
return f"missing required key {key!r}"
expected = (properties.get(key) or {}).get("type")
if expected and expected in type_map:
if not isinstance(payload[key], type_map[expected]):
return (
f"key {key!r} has wrong type: "
f"expected {expected}, got {type(payload[key]).__name__}"
)
# bool is a subclass of int in Python; reject `True`/`False`
# when the schema asked for an integer/number.
if expected in ("integer", "number") and isinstance(payload[key], bool):
return f"key {key!r} is a bool but schema asked for {expected}"
return None
def test_structured_outputs_azure(compat_result):
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
assert the trailing `result` event contains a schema-conforming
`structured_output`."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=["--json-schema", SCHEMA_JSON],
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
payload = _extract_structured_output(outcome.events)
if payload is None:
error = (
f"[{model}] no `structured_output` in trailing result event; "
"Claude Code's StructuredOutput tool round-trip did not "
"complete end-to-end through the proxy"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
shape_error = _validate_against_schema(payload, SCHEMA)
if shape_error is not None:
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,220 @@
"""structured_outputs x Bedrock (Converse).
Drive the real `claude` CLI in headless mode with the `--json-schema`
flag, route through a LiteLLM proxy aimed at Bedrock (Converse), and assert that
the final stream-json `result` event surfaces a `structured_output`
object whose shape matches the schema.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/structured_outputs/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
What this row actually exercises (and what it does not):
`--json-schema` is implemented client-side by Claude Code: the CLI
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
equals the user-supplied JSON Schema, forces the model toward it, and
finally extracts the tool_use input on the trailing `result` event as
`structured_output: {...}`. The proxy never sees `output_config.schema`
in this flow -- it sees a normal `tools` array with one synthetic
tool.
This makes the row a tool-use feature test in disguise. It's still a
distinct row from `tool_use` because:
- The synthetic tool is generated per request from a user schema, not
a developer-declared one. Provider-side bugs that special-case
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
blocks back to lowercase, or stripping the StructuredOutput-only
`additionalProperties: false`) only surface here.
- The success signal lives on the *final* `result` event, not the
intermediate `assistant` events the `tool_use` row checks. A proxy
that drops trailing events (seen in early Bedrock Converse SSE
plumbing) breaks this cell while leaving `tool_use` green.
It is NOT a test of Anthropic's server-side `output_config.schema`
parameter -- that's a different feature used internally by Claude Code
for session-title generation and is not reachable from any CLI flag.
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
the `count_tokens` and other HTTP-probe rows, not here.
Three Claude tiers run in parallel; one `compat_result.add(...)` per
tier so the matrix's "all three must pass" rule applies.
"""
from __future__ import annotations
import json
import os
import re
from typing import Any, Mapping, Optional, Sequence, Tuple
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
# Minimal schema with one required integer field. Kept intentionally
# small -- the matrix tests the *plumbing*, not the model's ability to
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
# integer schema gives every tier (including Haiku) enough headroom
# that schema satisfaction is essentially deterministic, isolating
# failures to the proxy / transport.
SCHEMA = {
"type": "object",
"properties": {"answer": {"type": "integer"}},
"required": ["answer"],
"additionalProperties": False,
}
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
# A prompt the model has no reason to misanswer; we don't check the
# value, but a wrong answer would suggest the structured-output
# pathway is silently degrading reasoning, which is itself worth
# noticing.
PROMPT = "What is 2 + 2? Reply only via the structured output."
def _extract_structured_output(
events: Sequence[Mapping[str, Any]],
) -> Optional[Mapping[str, Any]]:
"""Return the `structured_output` payload from the last `result` event.
Claude Code emits its terminal stream-json line as
`{"type":"result","structured_output":{...},...}` when a request
used `--json-schema` and the model actually produced a valid tool
call. If the model bailed or the proxy ate the trailing events,
`structured_output` is missing -- which is exactly the failure
mode we want this row to surface, so the caller treats `None` as
"feature did not work end-to-end".
"""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
so = event.get("structured_output")
if isinstance(so, Mapping):
return so
return None
def _validate_against_schema(
payload: Mapping[str, Any], schema: Mapping[str, Any]
) -> Optional[str]:
"""Tiny shape validator covering the subset we actually need.
We deliberately do not pull in `jsonschema` as a test dep: the
matrix's success signal is "does the proxy let the synthetic
StructuredOutput tool round-trip end-to-end", and that's
answerable with a presence + type check over `required` keys.
Any malformed schema beyond that would be a Claude Code bug,
not a LiteLLM-proxy bug, so a deeper check would only add false
failures on the wrong axis.
"""
type_map = {
"integer": int,
"number": (int, float),
"string": str,
"boolean": bool,
"array": list,
"object": Mapping,
}
required = schema.get("required") or []
properties = schema.get("properties") or {}
for key in required:
if key not in payload:
return f"missing required key {key!r}"
expected = (properties.get(key) or {}).get("type")
if expected and expected in type_map:
if not isinstance(payload[key], type_map[expected]):
return (
f"key {key!r} has wrong type: "
f"expected {expected}, got {type(payload[key]).__name__}"
)
# bool is a subclass of int in Python; reject `True`/`False`
# when the schema asked for an integer/number.
if expected in ("integer", "number") and isinstance(payload[key], bool):
return f"key {key!r} is a bool but schema asked for {expected}"
return None
def test_structured_outputs_bedrock_converse(compat_result):
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
assert the trailing `result` event contains a schema-conforming
`structured_output`."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt=PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=["--json-schema", SCHEMA_JSON],
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
payload = _extract_structured_output(outcome.events)
if payload is None:
error = (
f"[{model}] no `structured_output` in trailing result event; "
"Claude Code's StructuredOutput tool round-trip did not "
"complete end-to-end through the proxy"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
shape_error = _validate_against_schema(payload, SCHEMA)
if shape_error is not None:
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,220 @@
"""structured_outputs x Bedrock (Invoke).
Drive the real `claude` CLI in headless mode with the `--json-schema`
flag, route through a LiteLLM proxy aimed at Bedrock (Invoke), and assert that
the final stream-json `result` event surfaces a `structured_output`
object whose shape matches the schema.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/structured_outputs/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
What this row actually exercises (and what it does not):
`--json-schema` is implemented client-side by Claude Code: the CLI
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
equals the user-supplied JSON Schema, forces the model toward it, and
finally extracts the tool_use input on the trailing `result` event as
`structured_output: {...}`. The proxy never sees `output_config.schema`
in this flow -- it sees a normal `tools` array with one synthetic
tool.
This makes the row a tool-use feature test in disguise. It's still a
distinct row from `tool_use` because:
- The synthetic tool is generated per request from a user schema, not
a developer-declared one. Provider-side bugs that special-case
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
blocks back to lowercase, or stripping the StructuredOutput-only
`additionalProperties: false`) only surface here.
- The success signal lives on the *final* `result` event, not the
intermediate `assistant` events the `tool_use` row checks. A proxy
that drops trailing events (seen in early Bedrock Converse SSE
plumbing) breaks this cell while leaving `tool_use` green.
It is NOT a test of Anthropic's server-side `output_config.schema`
parameter -- that's a different feature used internally by Claude Code
for session-title generation and is not reachable from any CLI flag.
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
the `count_tokens` and other HTTP-probe rows, not here.
Three Claude tiers run in parallel; one `compat_result.add(...)` per
tier so the matrix's "all three must pass" rule applies.
"""
from __future__ import annotations
import json
import os
import re
from typing import Any, Mapping, Optional, Sequence, Tuple
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
# Minimal schema with one required integer field. Kept intentionally
# small -- the matrix tests the *plumbing*, not the model's ability to
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
# integer schema gives every tier (including Haiku) enough headroom
# that schema satisfaction is essentially deterministic, isolating
# failures to the proxy / transport.
SCHEMA = {
"type": "object",
"properties": {"answer": {"type": "integer"}},
"required": ["answer"],
"additionalProperties": False,
}
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
# A prompt the model has no reason to misanswer; we don't check the
# value, but a wrong answer would suggest the structured-output
# pathway is silently degrading reasoning, which is itself worth
# noticing.
PROMPT = "What is 2 + 2? Reply only via the structured output."
def _extract_structured_output(
events: Sequence[Mapping[str, Any]],
) -> Optional[Mapping[str, Any]]:
"""Return the `structured_output` payload from the last `result` event.
Claude Code emits its terminal stream-json line as
`{"type":"result","structured_output":{...},...}` when a request
used `--json-schema` and the model actually produced a valid tool
call. If the model bailed or the proxy ate the trailing events,
`structured_output` is missing -- which is exactly the failure
mode we want this row to surface, so the caller treats `None` as
"feature did not work end-to-end".
"""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
so = event.get("structured_output")
if isinstance(so, Mapping):
return so
return None
def _validate_against_schema(
payload: Mapping[str, Any], schema: Mapping[str, Any]
) -> Optional[str]:
"""Tiny shape validator covering the subset we actually need.
We deliberately do not pull in `jsonschema` as a test dep: the
matrix's success signal is "does the proxy let the synthetic
StructuredOutput tool round-trip end-to-end", and that's
answerable with a presence + type check over `required` keys.
Any malformed schema beyond that would be a Claude Code bug,
not a LiteLLM-proxy bug, so a deeper check would only add false
failures on the wrong axis.
"""
type_map = {
"integer": int,
"number": (int, float),
"string": str,
"boolean": bool,
"array": list,
"object": Mapping,
}
required = schema.get("required") or []
properties = schema.get("properties") or {}
for key in required:
if key not in payload:
return f"missing required key {key!r}"
expected = (properties.get(key) or {}).get("type")
if expected and expected in type_map:
if not isinstance(payload[key], type_map[expected]):
return (
f"key {key!r} has wrong type: "
f"expected {expected}, got {type(payload[key]).__name__}"
)
# bool is a subclass of int in Python; reject `True`/`False`
# when the schema asked for an integer/number.
if expected in ("integer", "number") and isinstance(payload[key], bool):
return f"key {key!r} is a bool but schema asked for {expected}"
return None
def test_structured_outputs_bedrock_invoke(compat_result):
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
assert the trailing `result` event contains a schema-conforming
`structured_output`."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=["--json-schema", SCHEMA_JSON],
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
payload = _extract_structured_output(outcome.events)
if payload is None:
error = (
f"[{model}] no `structured_output` in trailing result event; "
"Claude Code's StructuredOutput tool round-trip did not "
"complete end-to-end through the proxy"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
shape_error = _validate_against_schema(payload, SCHEMA)
if shape_error is not None:
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,220 @@
"""structured_outputs x Vertex AI.
Drive the real `claude` CLI in headless mode with the `--json-schema`
flag, route through a LiteLLM proxy aimed at Vertex AI, and assert that
the final stream-json `result` event surfaces a `structured_output`
object whose shape matches the schema.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/structured_outputs/test_vertex_ai.py
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
What this row actually exercises (and what it does not):
`--json-schema` is implemented client-side by Claude Code: the CLI
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
equals the user-supplied JSON Schema, forces the model toward it, and
finally extracts the tool_use input on the trailing `result` event as
`structured_output: {...}`. The proxy never sees `output_config.schema`
in this flow -- it sees a normal `tools` array with one synthetic
tool.
This makes the row a tool-use feature test in disguise. It's still a
distinct row from `tool_use` because:
- The synthetic tool is generated per request from a user schema, not
a developer-declared one. Provider-side bugs that special-case
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
blocks back to lowercase, or stripping the StructuredOutput-only
`additionalProperties: false`) only surface here.
- The success signal lives on the *final* `result` event, not the
intermediate `assistant` events the `tool_use` row checks. A proxy
that drops trailing events (seen in early Bedrock Converse SSE
plumbing) breaks this cell while leaving `tool_use` green.
It is NOT a test of Anthropic's server-side `output_config.schema`
parameter -- that's a different feature used internally by Claude Code
for session-title generation and is not reachable from any CLI flag.
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
the `count_tokens` and other HTTP-probe rows, not here.
Three Claude tiers run in parallel; one `compat_result.add(...)` per
tier so the matrix's "all three must pass" rule applies.
"""
from __future__ import annotations
import json
import os
import re
from typing import Any, Mapping, Optional, Sequence, Tuple
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
# Minimal schema with one required integer field. Kept intentionally
# small -- the matrix tests the *plumbing*, not the model's ability to
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
# integer schema gives every tier (including Haiku) enough headroom
# that schema satisfaction is essentially deterministic, isolating
# failures to the proxy / transport.
SCHEMA = {
"type": "object",
"properties": {"answer": {"type": "integer"}},
"required": ["answer"],
"additionalProperties": False,
}
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
# A prompt the model has no reason to misanswer; we don't check the
# value, but a wrong answer would suggest the structured-output
# pathway is silently degrading reasoning, which is itself worth
# noticing.
PROMPT = "What is 2 + 2? Reply only via the structured output."
def _extract_structured_output(
events: Sequence[Mapping[str, Any]],
) -> Optional[Mapping[str, Any]]:
"""Return the `structured_output` payload from the last `result` event.
Claude Code emits its terminal stream-json line as
`{"type":"result","structured_output":{...},...}` when a request
used `--json-schema` and the model actually produced a valid tool
call. If the model bailed or the proxy ate the trailing events,
`structured_output` is missing -- which is exactly the failure
mode we want this row to surface, so the caller treats `None` as
"feature did not work end-to-end".
"""
for event in reversed(list(events)):
if event.get("type") != "result":
continue
so = event.get("structured_output")
if isinstance(so, Mapping):
return so
return None
def _validate_against_schema(
payload: Mapping[str, Any], schema: Mapping[str, Any]
) -> Optional[str]:
"""Tiny shape validator covering the subset we actually need.
We deliberately do not pull in `jsonschema` as a test dep: the
matrix's success signal is "does the proxy let the synthetic
StructuredOutput tool round-trip end-to-end", and that's
answerable with a presence + type check over `required` keys.
Any malformed schema beyond that would be a Claude Code bug,
not a LiteLLM-proxy bug, so a deeper check would only add false
failures on the wrong axis.
"""
type_map = {
"integer": int,
"number": (int, float),
"string": str,
"boolean": bool,
"array": list,
"object": Mapping,
}
required = schema.get("required") or []
properties = schema.get("properties") or {}
for key in required:
if key not in payload:
return f"missing required key {key!r}"
expected = (properties.get(key) or {}).get("type")
if expected and expected in type_map:
if not isinstance(payload[key], type_map[expected]):
return (
f"key {key!r} has wrong type: "
f"expected {expected}, got {type(payload[key]).__name__}"
)
# bool is a subclass of int in Python; reject `True`/`False`
# when the schema asked for an integer/number.
if expected in ("integer", "number") and isinstance(payload[key], bool):
return f"key {key!r} is a bool but schema asked for {expected}"
return None
def test_structured_outputs_vertex_ai(compat_result):
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
assert the trailing `result` event contains a schema-conforming
`structured_output`."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
pytrace=False,
)
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt=PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=["--json-schema", SCHEMA_JSON],
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
payload = _extract_structured_output(outcome.events)
if payload is None:
error = (
f"[{model}] no `structured_output` in trailing result event; "
"Claude Code's StructuredOutput tool round-trip did not "
"complete end-to-end through the proxy"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
shape_error = _validate_against_schema(payload, SCHEMA)
if shape_error is not None:
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,103 @@
# Proxy routing config for the Claude Code Compatibility Matrix PR gate.
#
# The tests under `tests/e2e/claude_code/` only know **alias** names (e.g.
# `claude-haiku-4-5-bedrock-invoke`). The proxy is the layer that maps
# each alias to a real upstream model id, region, and credentials.
#
# Adding a new (feature, provider) cell is a three-step change in the
# test repo (manifest + test file + alias here); changing which upstream
# model a cell exercises is a one-step change here, with no test edits.
#
# Aliases:
# - claude-{tier} → Anthropic API
# - claude-{tier}-bedrock-invoke → Bedrock InvokeModel API
# - claude-{tier}-bedrock-converse → Bedrock Converse API
# - claude-{tier}-vertex → GCP Vertex AI
# - claude-{tier}-azure → Microsoft Foundry (Anthropic deployments)
model_list:
# ---- Anthropic ----
- model_name: claude-haiku-4-5
litellm_params:
model: anthropic/claude-haiku-4-5
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-sonnet-4-6
litellm_params:
model: anthropic/claude-sonnet-4-6
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: claude-opus-4-7
litellm_params:
model: anthropic/claude-opus-4-7
api_key: os.environ/ANTHROPIC_API_KEY
# ---- Bedrock (InvokeModel) ----
- model_name: claude-haiku-4-5-bedrock-invoke
litellm_params:
model: bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0
aws_region_name: us-east-1
- model_name: claude-sonnet-4-6-bedrock-invoke
litellm_params:
model: bedrock/us.anthropic.claude-sonnet-4-6
aws_region_name: us-east-1
- model_name: claude-opus-4-7-bedrock-invoke
litellm_params:
model: bedrock/us.anthropic.claude-opus-4-7
aws_region_name: us-east-1
# ---- Bedrock (Converse) ----
- model_name: claude-haiku-4-5-bedrock-converse
litellm_params:
model: bedrock/converse/us.anthropic.claude-haiku-4-5-20251001-v1:0
aws_region_name: us-east-1
- model_name: claude-sonnet-4-6-bedrock-converse
litellm_params:
model: bedrock/converse/us.anthropic.claude-sonnet-4-6
aws_region_name: us-east-1
- model_name: claude-opus-4-7-bedrock-converse
litellm_params:
model: bedrock/converse/us.anthropic.claude-opus-4-7
aws_region_name: us-east-1
# ---- Vertex AI ----
- model_name: claude-haiku-4-5-vertex
litellm_params:
model: vertex_ai/claude-haiku-4-5
vertex_ai_project: os.environ/VERTEXAI_PROJECT
vertex_ai_location: os.environ/VERTEXAI_LOCATION
- model_name: claude-sonnet-4-6-vertex
litellm_params:
model: vertex_ai/claude-sonnet-4-6
vertex_ai_project: os.environ/VERTEXAI_PROJECT
vertex_ai_location: os.environ/VERTEXAI_LOCATION
- model_name: claude-opus-4-7-vertex
litellm_params:
model: vertex_ai/claude-opus-4-7
vertex_ai_project: os.environ/VERTEXAI_PROJECT
vertex_ai_location: os.environ/VERTEXAI_LOCATION
# ---- Microsoft Foundry (Anthropic deployments on Azure) ----
- model_name: claude-haiku-4-5-azure
litellm_params:
model: azure_ai/claude-haiku-4-5
api_base: os.environ/AZURE_FOUNDRY_API_BASE
api_key: os.environ/AZURE_FOUNDRY_API_KEY
- model_name: claude-sonnet-4-6-azure
litellm_params:
model: azure_ai/claude-sonnet-4-6
api_base: os.environ/AZURE_FOUNDRY_API_BASE
api_key: os.environ/AZURE_FOUNDRY_API_KEY
- model_name: claude-opus-4-7-azure
litellm_params:
model: azure_ai/claude-opus-4-7
api_base: os.environ/AZURE_FOUNDRY_API_BASE
api_key: os.environ/AZURE_FOUNDRY_API_KEY
general_settings:
# Claude Code sends provider-specific headers (e.g. anthropic-beta) we
# want to forward verbatim to the upstream so the wire-shape under
# test matches what real customers send.
forward_client_headers_to_llm_api: true
litellm_settings:
drop_params: true
modify_params: true

View file

@ -0,0 +1,132 @@
"""thinking x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, enable extended thinking via `--effort high`, and assert
that the upstream returned a `thinking` content block. This proves the
proxy preserves Anthropic's `thinking` request parameter and the
upstream response's `thinking` content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking/test_anthropic.py
^^^^^^^^ ^^^^^^^^^
feature_id provider
The three Claude tiers run in parallel inside this single test, with
one `compat_result.add(...)` entry per model so the matrix builder
still sees three rows for this (feature, provider).
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
# --effort max maps to the largest thinking budget on every supported
# Claude tier; the test cares about wire shape, not answer quality. We
# use a CLI flag (rather than the legacy MAX_THINKING_TOKENS env var)
# because Claude Code 2.x reads thinking config from --effort, not from
# the env, and silently no-ops the env var. We use `max` rather than
# `high` because Sonnet 4.6 / Opus 4.7 only emit thinking blocks when
# the budget is generous and the prompt is non-trivial.
THINKING_ARGS = ["--effort", "max"]
# A puzzle non-trivial enough that Sonnet/Opus actually engage thinking
# rather than answer from memory. Trivial arithmetic ("3-2=?") is
# optimized away on the modern tiers and arrives without a thinking
# block, which would make this test silently false-fail under
# `--effort max`. Haiku 4.5 thinks even for trivial prompts; Sonnet 4.6
# and Opus 4.7 only emit thinking when the upstream judges it useful.
THINKING_PROMPT = (
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a `thinking` content block."""
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "thinking":
return True
return False
def test_thinking_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and assert a `thinking` content block was emitted."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=THINKING_ARGS,
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_thinking_block(outcome.events):
error = (
f"[{model}] no `thinking` content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,120 @@
"""thinking x Azure (Microsoft Foundry).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models hosted in Microsoft Foundry on
Azure, enable extended thinking via `--effort high`, and assert
that the upstream returned a `thinking` content block.
Foundry's Claude deployments advertise `supports_reasoning: true` in
LiteLLM's pricing metadata; the `thinking={"type": "enabled", ...}`
parameter passes through `azure_ai/claude-*` to Foundry's
`/anthropic/v1/messages` endpoint unchanged. Note that
`claude-opus-4-7-preview` documents thinking as not supported on
Foundry; if that lands, this row may flip to a partial pass and we'll
re-evaluate.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking/test_azure.py
^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "thinking":
return True
return False
def test_thinking_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and assert a `thinking` content block was emitted."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=THINKING_ARGS,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_thinking_block(outcome.events):
error = (
f"[{model}] no `thinking` content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,112 @@
"""thinking x Bedrock (Converse).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the unified `Converse` API path,
enable extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking/test_bedrock_converse.py
^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "thinking":
return True
return False
def test_thinking_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and assert a `thinking` content block was emitted."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=THINKING_ARGS,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_thinking_block(outcome.events):
error = (
f"[{model}] no `thinking` content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,112 @@
"""thinking x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
enable extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking/test_bedrock_invoke.py
^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "thinking":
return True
return False
def test_thinking_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and assert a `thinking` content block was emitted."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=THINKING_ARGS,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_thinking_block(outcome.events):
error = (
f"[{model}] no `thinking` content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,112 @@
"""thinking x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models on Google Cloud Vertex AI, enable
extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking/test_vertex_ai.py
^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == "thinking":
return True
return False
def test_thinking_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and assert a `thinking` content block was emitted."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_args=THINKING_ARGS,
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_thinking_block(outcome.events):
error = (
f"[{model}] no `thinking` content block observed in stream-json events"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,158 @@
"""thinking_with_tool_use x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, enable extended thinking via `--effort high`, allow
the built-in `Bash` tool, and ask Claude to plan-and-execute a task
that requires both reasoning and a tool call. Assert the upstream
returned both a `thinking` content block and a `tool_use` content
block in the same turn proving the proxy preserves the wire shape
where extended thinking and tool use coexist.
This is the cell that historically catches the most provider bugs:
"thinking blocks cannot be modified" 400s, the recurring Bedrock
"thinking.type.enabled is not supported" error, and the
`fine-grained-tool-streaming` + `interleaved-thinking` beta-header
interactions. A regression in any of those collapses this cell.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking_with_tool_use/test_anthropic.py
^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
ANTHROPIC_MODELS = [
"claude-haiku-4-5",
"claude-sonnet-4-6",
"claude-opus-4-7",
]
# Extended thinking on, with a small budget — enough to surface a
# non-empty thinking block on a trivial reasoning prompt without
# blowing up wall time.
THINKING_ARGS = ["--effort", "max"]
# Prompt designed to force both blocks: the model has to *reason* about
# what command to run before *invoking* the Bash tool. Using a fixed
# expected output keeps the assertion focused on the wire shape rather
# than on answer quality.
# Prompt fixes the exact bash command to `echo pong`. The thinking
# block is preserved (the model reasons about why `echo pong` works),
# but the executed command is pinned so the cell can run under the
# tight `Bash(echo pong) + dontAsk` permission below — see
# `tool_use/test_anthropic.py` for the full security rationale.
THINKING_TOOL_PROMPT = (
"Think step by step about why the command `echo pong` prints just the "
"word 'pong'. Then use the Bash tool to run exactly the command "
"`echo pong` and report what it printed."
)
TOOL_USE_ARGS = [
"--allowed-tools",
"Bash(echo pong)",
"--permission-mode",
"dontAsk",
]
def _has_block_type(
events: Sequence[Mapping[str, Any]],
block_type: str,
) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a content block of the given type."""
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == block_type:
return True
return False
def test_thinking_with_tool_use_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
enabled and tool use, and assert both `thinking` and `tool_use`
content blocks landed in the same turn."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []
for model in ANTHROPIC_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "thinking"):
error = (
f"[{model}] no `thinking` content block observed; thinking "
f"either disabled by the proxy or stripped by the upstream"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "tool_use"):
error = (
f"[{model}] no `tool_use` content block observed alongside "
f"thinking; the proxy may have dropped tools when thinking is on"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,130 @@
"""thinking_with_tool_use x Microsoft Foundry (Azure).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
enable extended thinking via `--effort high`, allow the built-in
`Bash` tool, and ask Claude to plan-and-execute a task that requires
both reasoning and a tool call. Assert the upstream returned both a
`thinking` content block and a `tool_use` content block in the same
turn.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking_with_tool_use/test_azure.py
^^^^^^^^^^^^^^^^^^^^^^ ^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
AZURE_MODELS = [
"claude-haiku-4-5-azure",
"claude-sonnet-4-6-azure",
"claude-opus-4-7-azure",
]
THINKING_ARGS = ["--effort", "max"]
# Prompt + Bash restriction pin the executed command to `echo pong`;
# see `tool_use/test_anthropic.py` for the security rationale.
THINKING_TOOL_PROMPT = (
"Think step by step about why the command `echo pong` prints just the "
"word 'pong'. Then use the Bash tool to run exactly the command "
"`echo pong` and report what it printed."
)
TOOL_USE_ARGS = [
"--allowed-tools",
"Bash(echo pong)",
"--permission-mode",
"dontAsk",
]
def _has_block_type(
events: Sequence[Mapping[str, Any]],
block_type: str,
) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == block_type:
return True
return False
def test_thinking_with_tool_use_azure(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []
for model in AZURE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "thinking"):
error = f"[{model}] no `thinking` content block observed"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "tool_use"):
error = f"[{model}] no `tool_use` content block observed alongside thinking"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,135 @@
"""thinking_with_tool_use x Bedrock (Converse).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the `Converse` API, enable extended
thinking via `--effort high`, allow the built-in `Bash` tool, and
ask Claude to plan-and-execute a task that requires both reasoning and
a tool call. Assert the upstream returned both a `thinking` content
block and a `tool_use` content block in the same turn.
The Converse API has its own `additionalModelRequestFields.thinking`
shape and its own tool-use envelope; this cell catches gateway
regressions where the proxy fails to translate between Anthropic's
`thinking` parameter and Converse's reasoning configuration when tools
are also present.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_converse.py
^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_CONVERSE_MODELS = [
"claude-haiku-4-5-bedrock-converse",
"claude-sonnet-4-6-bedrock-converse",
"claude-opus-4-7-bedrock-converse",
]
THINKING_ARGS = ["--effort", "max"]
# Prompt + Bash restriction pin the executed command to `echo pong`;
# see `tool_use/test_anthropic.py` for the security rationale.
THINKING_TOOL_PROMPT = (
"Think step by step about why the command `echo pong` prints just the "
"word 'pong'. Then use the Bash tool to run exactly the command "
"`echo pong` and report what it printed."
)
TOOL_USE_ARGS = [
"--allowed-tools",
"Bash(echo pong)",
"--permission-mode",
"dontAsk",
]
def _has_block_type(
events: Sequence[Mapping[str, Any]],
block_type: str,
) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == block_type:
return True
return False
def test_thinking_with_tool_use_bedrock_converse(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []
for model in BEDROCK_CONVERSE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "thinking"):
error = f"[{model}] no `thinking` content block observed"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "tool_use"):
error = f"[{model}] no `tool_use` content block observed alongside thinking"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,137 @@
"""thinking_with_tool_use x Bedrock (Invoke).
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
enable extended thinking via `--effort high`, allow the built-in
`Bash` tool, and ask Claude to plan-and-execute a task that requires
both reasoning and a tool call. Assert the upstream returned both a
`thinking` content block and a `tool_use` content block in the same
turn.
This is the cell most likely to flush out Bedrock-specific bugs in
LiteLLM's Anthropic <-> Bedrock translation: the recurring
"thinking.type.enabled is not supported" 400 error has reappeared on
several Bedrock model routes (notably application inference profile
ARNs), and the only reliable signal that the fix is wired through the
proxy is a successful round-trip on this cell.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking_with_tool_use/test_bedrock_invoke.py
^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
BEDROCK_INVOKE_MODELS = [
"claude-haiku-4-5-bedrock-invoke",
"claude-sonnet-4-6-bedrock-invoke",
"claude-opus-4-7-bedrock-invoke",
]
THINKING_ARGS = ["--effort", "max"]
# Prompt + Bash restriction pin the executed command to `echo pong`;
# see `tool_use/test_anthropic.py` for the security rationale.
THINKING_TOOL_PROMPT = (
"Think step by step about why the command `echo pong` prints just the "
"word 'pong'. Then use the Bash tool to run exactly the command "
"`echo pong` and report what it printed."
)
TOOL_USE_ARGS = [
"--allowed-tools",
"Bash(echo pong)",
"--permission-mode",
"dontAsk",
]
def _has_block_type(
events: Sequence[Mapping[str, Any]],
block_type: str,
) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == block_type:
return True
return False
def test_thinking_with_tool_use_bedrock_invoke(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []
for model in BEDROCK_INVOKE_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "thinking"):
error = f"[{model}] no `thinking` content block observed"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "tool_use"):
error = f"[{model}] no `tool_use` content block observed alongside thinking"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

View file

@ -0,0 +1,135 @@
"""thinking_with_tool_use x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to GCP Vertex AI, enable extended thinking via
`--effort high`, allow the built-in `Bash` tool, and ask Claude
to plan-and-execute a task that requires both reasoning and a tool
call. Assert the upstream returned both a `thinking` content block and
a `tool_use` content block in the same turn.
Vertex AI exposes Anthropic models via `:rawPredict` /
`:streamRawPredict` and has its own beta-header allowlist. This cell
catches gateway regressions where the proxy strips
`anthropic-beta: interleaved-thinking-2025-05-14` (or the equivalent
header set the upstream needs) on the way to Vertex.
The (feature, provider) for this cell is inferred from the file path by
`tests/e2e/claude_code/conftest.py`:
tests/e2e/claude_code/thinking_with_tool_use/test_vertex_ai.py
^^^^^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
from __future__ import annotations
import os
from typing import Any, Mapping, Sequence
import pytest
from claude_code.cli_driver import (
ClaudeCLIError,
failure_diagnostic,
run_claude_models_parallel,
)
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
VERTEX_AI_MODELS = [
"claude-haiku-4-5-vertex",
"claude-sonnet-4-6-vertex",
"claude-opus-4-7-vertex",
]
THINKING_ARGS = ["--effort", "max"]
# Prompt + Bash restriction pin the executed command to `echo pong`;
# see `tool_use/test_anthropic.py` for the security rationale.
THINKING_TOOL_PROMPT = (
"Think step by step about why the command `echo pong` prints just the "
"word 'pong'. Then use the Bash tool to run exactly the command "
"`echo pong` and report what it printed."
)
TOOL_USE_ARGS = [
"--allowed-tools",
"Bash(echo pong)",
"--permission-mode",
"dontAsk",
]
def _has_block_type(
events: Sequence[Mapping[str, Any]],
block_type: str,
) -> bool:
for event in events:
if event.get("type") != "assistant":
continue
message = event.get("message") or {}
content = message.get("content")
if not isinstance(content, list):
continue
for block in content:
if isinstance(block, dict) and block.get("type") == block_type:
return True
return False
def test_thinking_with_tool_use_vertex_ai(compat_result):
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
compat_result.set(
{
"status": "fail",
"error": (
f"missing required env: set {PROXY_BASE_URL_ENV} and "
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
),
}
)
pytest.fail(
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []
for model in VERTEX_AI_MODELS:
outcome = outcomes[model]
if isinstance(outcome, ClaudeCLIError):
error = f"[{model}] {outcome}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if outcome.exit_code != 0:
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "thinking"):
error = f"[{model}] no `thinking` content block observed"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
if not _has_block_type(outcome.events, "tool_use"):
error = f"[{model}] no `tool_use` content block observed alongside thinking"
compat_result.add({"status": "fail", "error": error})
failures.append(error)
continue
compat_result.add({"status": "pass"})
if failures:
pytest.fail("; ".join(failures), pytrace=False)

Some files were not shown because too many files have changed in this diff Show more