mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(claude_code): rename thinking row + add 4 feature rows (15 total)
Matrix grows from 11 to 15 feature rows. All new tests collected + 180 unit tests still pass; smoke runs hit real LiteLLM bug surfaces on bedrock_invoke, bedrock_converse, and vertex_ai (cells correctly red in PR #142). Rename ------ `extended_thinking` -> `thinking` (directory, manifest id+name, 5 test fn names, 5 docstrings, builder unit-test fixtures, sample JSON, run_compat.sh). Existing test logic already covers both manual (`thinking.type=enabled`, Haiku 4.5) and adaptive (`thinking.type=adaptive`, Opus 4.7) shapes because Claude Code picks the shape per model from `--effort max`; the name change just stops the column from looking like a Claude 3.7 reference. New rows -------- - structured_outputs (5 files, CLI `--json-schema`). Claude Code synthesizes a single `StructuredOutput` tool from the schema and surfaces the tool_use input as `structured_output` on the trailing `result` event. Test ships its own `_validate_against_schema` so we don't take a jsonschema dep just for matrix surface. - count_tokens (5 files, HTTP probe). POSTs the proxy's `/v1/messages/count_tokens` directly and asserts the response is `{input_tokens: positive int}`. No CLI hook exists for this endpoint; the test goes through the new http_probe helper instead. - tool_search (5 files, HTTP probe). Sends `tools: [{type: tool_search_tool_regex_20251119, name: tool_search_tool_regex}]` and asserts the proxy doesn't 400. MCP fan-out via `--mcp-config` would also exercise the tool-search beta header path, but it's flaky w.r.t. Claude Code's internal tool-deferral threshold; the HTTP probe hits the actual bug surface (per-provider beta-header translation `advanced-tool-use-2025-11-20` vs `tool-search-tool-2025-10-19`). - long_context_1m (5 files, CLI `--betas context-1m-2025-08-07 --max-budget-usd 6`). A ~210k-token padded prompt over stdin exercises the 1M-context beta. Sonnet 4.6 + Opus 4.7 only -- Haiku 4.5's window is 200k, so it's excluded from MODELS (not marked not_applicable) to keep the per-cell aggregator semantics intact. Prompt uses a document-style preamble + 8 cycling pangrams rather than repeating identical chunks; without that, Opus 4.7 trips the safety filter mid-response with a Usage Policy refusal. `--max-budget-usd 6` is a runaway-loop guard, ~2x worst-case Opus per-cell spend. New helper ---------- `tests/claude_code/http_probe.py`: shared `ProbeResult` dataclass plus per-endpoint `probe_*` + `assert_*_shape` pairs for the HTTP-probe rows. Uses httpx with `anthropic-version: 2023-06-01` and a 30s timeout.
This commit is contained in:
parent
891da2372d
commit
be65b4e23b
36 changed files with 3550 additions and 32 deletions
|
|
@ -206,7 +206,12 @@ def test_build_matrix_6x5_grid_matches_published_sample():
|
|||
"tool_use",
|
||||
"prompt_caching_5m",
|
||||
"vision",
|
||||
"extended_thinking",
|
||||
# Row 6 of the v0 PRD; originally shipped as `extended_thinking`.
|
||||
# The id was renamed in-place to `thinking` to match Anthropic's
|
||||
# current docs (which reserve "extended thinking" for the
|
||||
# deprecated manual mode only). The row's *position* in v0 is
|
||||
# the load-bearing invariant, not the id string.
|
||||
"thinking",
|
||||
]
|
||||
v0_features = [
|
||||
feature
|
||||
|
|
|
|||
|
|
@ -26,7 +26,13 @@ EXPECTED_FEATURE_IDS = [
|
|||
"tool_use",
|
||||
"prompt_caching_5m",
|
||||
"vision",
|
||||
"extended_thinking",
|
||||
# v0 originally shipped this row as `extended_thinking`. It was
|
||||
# renamed in-place to `thinking` because Anthropic's docs reserve
|
||||
# "extended thinking" for the deprecated manual API mode only; the
|
||||
# single row exercises both manual and adaptive shapes since Claude
|
||||
# Code picks per model. The PRD's "v0" identity is the *position*
|
||||
# (row 6, 0-indexed 5), not the id string.
|
||||
"thinking",
|
||||
]
|
||||
|
||||
# The PRD's column order. Every feature directory must have one
|
||||
|
|
|
|||
94
tests/claude_code/count_tokens/test_anthropic.py
Normal file
94
tests/claude_code/count_tokens/test_anthropic.py
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
"""count_tokens x Anthropic.
|
||||
|
||||
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
|
||||
the `claude` CLI: it `POST`s directly to
|
||||
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
|
||||
the response is shaped `{"input_tokens": <positive int>}`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/count_tokens/test_anthropic.py
|
||||
^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI:
|
||||
|
||||
Claude Code calls `count_tokens` internally to compute budget /
|
||||
context-window usage display, but the result is consumed by the CLI
|
||||
in-process and never appears in stream-json events. There is no CLI
|
||||
flag that emits the count to stdout in a way our existing
|
||||
stream-json parser can pick up, so we can't test the endpoint round
|
||||
trip through the CLI surface.
|
||||
|
||||
The proxy *is* expected to expose `/v1/messages/count_tokens` for
|
||||
every Claude-style provider it routes to -- LiteLLM has historically
|
||||
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
|
||||
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
|
||||
Treating it as a matrix row keeps regressions in the cron's daily
|
||||
diff.
|
||||
|
||||
The cell goes red if *any* tier's probe fails the minimal shape
|
||||
check; the matrix's per-cell aggregator handles that automatically.
|
||||
Three tiers run sequentially because count_tokens is cheap (<100ms
|
||||
per request typical) and the parallelization that matters for the
|
||||
CLI rows isn't useful here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_count_tokens_shape,
|
||||
probe_count_tokens,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
ANTHROPIC_MODELS = [
|
||||
"claude-haiku-4-5",
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
|
||||
def test_count_tokens_anthropic(compat_result):
|
||||
"""Probe `/v1/messages/count_tokens` for each Anthropic tier and
|
||||
assert the response shape."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in ANTHROPIC_MODELS:
|
||||
result = probe_count_tokens(
|
||||
base_url=base_url, api_key=api_key, model=model
|
||||
)
|
||||
shape_error = assert_count_tokens_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] count_tokens probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
94
tests/claude_code/count_tokens/test_azure.py
Normal file
94
tests/claude_code/count_tokens/test_azure.py
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
"""count_tokens x Azure (Microsoft Foundry).
|
||||
|
||||
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
|
||||
the `claude` CLI: it `POST`s directly to
|
||||
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
|
||||
the response is shaped `{"input_tokens": <positive int>}`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/count_tokens/test_azure.py
|
||||
^^^^^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI:
|
||||
|
||||
Claude Code calls `count_tokens` internally to compute budget /
|
||||
context-window usage display, but the result is consumed by the CLI
|
||||
in-process and never appears in stream-json events. There is no CLI
|
||||
flag that emits the count to stdout in a way our existing
|
||||
stream-json parser can pick up, so we can't test the endpoint round
|
||||
trip through the CLI surface.
|
||||
|
||||
The proxy *is* expected to expose `/v1/messages/count_tokens` for
|
||||
every Claude-style provider it routes to -- LiteLLM has historically
|
||||
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
|
||||
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
|
||||
Treating it as a matrix row keeps regressions in the cron's daily
|
||||
diff.
|
||||
|
||||
The cell goes red if *any* tier's probe fails the minimal shape
|
||||
check; the matrix's per-cell aggregator handles that automatically.
|
||||
Three tiers run sequentially because count_tokens is cheap (<100ms
|
||||
per request typical) and the parallelization that matters for the
|
||||
CLI rows isn't useful here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_count_tokens_shape,
|
||||
probe_count_tokens,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
AZURE_MODELS = [
|
||||
"claude-haiku-4-5-azure",
|
||||
"claude-sonnet-4-6-azure",
|
||||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
|
||||
def test_count_tokens_azure(compat_result):
|
||||
"""Probe `/v1/messages/count_tokens` for each Azure (Microsoft Foundry) tier and
|
||||
assert the response shape."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in AZURE_MODELS:
|
||||
result = probe_count_tokens(
|
||||
base_url=base_url, api_key=api_key, model=model
|
||||
)
|
||||
shape_error = assert_count_tokens_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] count_tokens probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
94
tests/claude_code/count_tokens/test_bedrock_converse.py
Normal file
94
tests/claude_code/count_tokens/test_bedrock_converse.py
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
"""count_tokens x Bedrock (Converse).
|
||||
|
||||
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
|
||||
the `claude` CLI: it `POST`s directly to
|
||||
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
|
||||
the response is shaped `{"input_tokens": <positive int>}`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/count_tokens/test_bedrock_converse.py
|
||||
^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI:
|
||||
|
||||
Claude Code calls `count_tokens` internally to compute budget /
|
||||
context-window usage display, but the result is consumed by the CLI
|
||||
in-process and never appears in stream-json events. There is no CLI
|
||||
flag that emits the count to stdout in a way our existing
|
||||
stream-json parser can pick up, so we can't test the endpoint round
|
||||
trip through the CLI surface.
|
||||
|
||||
The proxy *is* expected to expose `/v1/messages/count_tokens` for
|
||||
every Claude-style provider it routes to -- LiteLLM has historically
|
||||
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
|
||||
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
|
||||
Treating it as a matrix row keeps regressions in the cron's daily
|
||||
diff.
|
||||
|
||||
The cell goes red if *any* tier's probe fails the minimal shape
|
||||
check; the matrix's per-cell aggregator handles that automatically.
|
||||
Three tiers run sequentially because count_tokens is cheap (<100ms
|
||||
per request typical) and the parallelization that matters for the
|
||||
CLI rows isn't useful here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_count_tokens_shape,
|
||||
probe_count_tokens,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_CONVERSE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-converse",
|
||||
"claude-sonnet-4-6-bedrock-converse",
|
||||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
|
||||
def test_count_tokens_bedrock_converse(compat_result):
|
||||
"""Probe `/v1/messages/count_tokens` for each Bedrock (Converse) tier and
|
||||
assert the response shape."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_CONVERSE_MODELS:
|
||||
result = probe_count_tokens(
|
||||
base_url=base_url, api_key=api_key, model=model
|
||||
)
|
||||
shape_error = assert_count_tokens_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] count_tokens probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
94
tests/claude_code/count_tokens/test_bedrock_invoke.py
Normal file
94
tests/claude_code/count_tokens/test_bedrock_invoke.py
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
"""count_tokens x Bedrock (Invoke).
|
||||
|
||||
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
|
||||
the `claude` CLI: it `POST`s directly to
|
||||
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
|
||||
the response is shaped `{"input_tokens": <positive int>}`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/count_tokens/test_bedrock_invoke.py
|
||||
^^^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI:
|
||||
|
||||
Claude Code calls `count_tokens` internally to compute budget /
|
||||
context-window usage display, but the result is consumed by the CLI
|
||||
in-process and never appears in stream-json events. There is no CLI
|
||||
flag that emits the count to stdout in a way our existing
|
||||
stream-json parser can pick up, so we can't test the endpoint round
|
||||
trip through the CLI surface.
|
||||
|
||||
The proxy *is* expected to expose `/v1/messages/count_tokens` for
|
||||
every Claude-style provider it routes to -- LiteLLM has historically
|
||||
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
|
||||
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
|
||||
Treating it as a matrix row keeps regressions in the cron's daily
|
||||
diff.
|
||||
|
||||
The cell goes red if *any* tier's probe fails the minimal shape
|
||||
check; the matrix's per-cell aggregator handles that automatically.
|
||||
Three tiers run sequentially because count_tokens is cheap (<100ms
|
||||
per request typical) and the parallelization that matters for the
|
||||
CLI rows isn't useful here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_count_tokens_shape,
|
||||
probe_count_tokens,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_INVOKE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-invoke",
|
||||
"claude-sonnet-4-6-bedrock-invoke",
|
||||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
|
||||
def test_count_tokens_bedrock_invoke(compat_result):
|
||||
"""Probe `/v1/messages/count_tokens` for each Bedrock (Invoke) tier and
|
||||
assert the response shape."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_INVOKE_MODELS:
|
||||
result = probe_count_tokens(
|
||||
base_url=base_url, api_key=api_key, model=model
|
||||
)
|
||||
shape_error = assert_count_tokens_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] count_tokens probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
94
tests/claude_code/count_tokens/test_vertex_ai.py
Normal file
94
tests/claude_code/count_tokens/test_vertex_ai.py
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
"""count_tokens x Vertex AI.
|
||||
|
||||
HTTP-probe row. Unlike the CLI-driven rows, this test never invokes
|
||||
the `claude` CLI: it `POST`s directly to
|
||||
`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts
|
||||
the response is shaped `{"input_tokens": <positive int>}`.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/count_tokens/test_vertex_ai.py
|
||||
^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI:
|
||||
|
||||
Claude Code calls `count_tokens` internally to compute budget /
|
||||
context-window usage display, but the result is consumed by the CLI
|
||||
in-process and never appears in stream-json events. There is no CLI
|
||||
flag that emits the count to stdout in a way our existing
|
||||
stream-json parser can pick up, so we can't test the endpoint round
|
||||
trip through the CLI surface.
|
||||
|
||||
The proxy *is* expected to expose `/v1/messages/count_tokens` for
|
||||
every Claude-style provider it routes to -- LiteLLM has historically
|
||||
had provider-specific bugs in this endpoint (Vertex AI `count_tokens`
|
||||
returned 400 to proxy gateways; see Claude Code release notes 2.1.121).
|
||||
Treating it as a matrix row keeps regressions in the cron's daily
|
||||
diff.
|
||||
|
||||
The cell goes red if *any* tier's probe fails the minimal shape
|
||||
check; the matrix's per-cell aggregator handles that automatically.
|
||||
Three tiers run sequentially because count_tokens is cheap (<100ms
|
||||
per request typical) and the parallelization that matters for the
|
||||
CLI rows isn't useful here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_count_tokens_shape,
|
||||
probe_count_tokens,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
VERTEX_AI_MODELS = [
|
||||
"claude-haiku-4-5-vertex",
|
||||
"claude-sonnet-4-6-vertex",
|
||||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
|
||||
def test_count_tokens_vertex_ai(compat_result):
|
||||
"""Probe `/v1/messages/count_tokens` for each Vertex AI tier and
|
||||
assert the response shape."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in VERTEX_AI_MODELS:
|
||||
result = probe_count_tokens(
|
||||
base_url=base_url, api_key=api_key, model=model
|
||||
)
|
||||
shape_error = assert_count_tokens_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] count_tokens probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
265
tests/claude_code/http_probe.py
Normal file
265
tests/claude_code/http_probe.py
Normal file
|
|
@ -0,0 +1,265 @@
|
|||
"""Direct HTTP probe helpers for the Claude Code compatibility matrix.
|
||||
|
||||
Most matrix cells drive the `claude` CLI in headless mode and observe
|
||||
the stream-json wire (see `cli_driver.py`). A handful of features the
|
||||
proxy must support don't have any CLI surface area -- `count_tokens` is
|
||||
the canonical example: Claude Code calls it internally for budget
|
||||
display, but the result never appears in stream-json events, so a CLI
|
||||
test cannot observe whether the endpoint round-tripped correctly
|
||||
through the proxy for any given provider.
|
||||
|
||||
This module is the second test pattern the matrix supports: a plain
|
||||
HTTP POST against a LiteLLM proxy endpoint, parsed and shape-checked
|
||||
in the test, with the same `compat_result` recording convention as the
|
||||
CLI-driven cells. The goal is to keep this pattern *narrow* -- if a
|
||||
feature can be tested via the CLI, it should be, because the CLI path
|
||||
is closer to what real Claude Code users hit. HTTP probes are only for
|
||||
features the CLI can't reach.
|
||||
|
||||
The probe deliberately uses a short timeout (30s) and small payloads:
|
||||
this is a "did the request shape survive the proxy's
|
||||
provider-specific transformations" test, not a load test, and a real
|
||||
endpoint regression typically surfaces in well under a second of wall
|
||||
time (400 / 500 from the upstream, or LiteLLM 500 on a transformation
|
||||
bug).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Mapping, Optional
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
DEFAULT_TIMEOUT_SECONDS = 30.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProbeResult:
|
||||
"""Structured outcome of a single HTTP probe.
|
||||
|
||||
`status_code` and `body` are the wire response; `payload` is the
|
||||
parsed JSON body if the response was JSON, else None. Tests assert
|
||||
on `status_code` + `payload` shape; `body` is preserved so failure
|
||||
diagnostics can echo the raw error string (which is the only thing
|
||||
a maintainer needs to triage a red cell).
|
||||
"""
|
||||
|
||||
status_code: int
|
||||
body: str
|
||||
payload: Optional[Mapping[str, Any]] = None
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
def probe_count_tokens(
|
||||
*,
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
model: str,
|
||||
message: str = "hello world",
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
) -> ProbeResult:
|
||||
"""POST to `{base_url}/v1/messages/count_tokens` for `model` and return the parsed result.
|
||||
|
||||
The Anthropic / LiteLLM `count_tokens` endpoint accepts a request
|
||||
body whose shape mirrors `/v1/messages` (model + messages), and
|
||||
returns `{"input_tokens": N}` for a successful response. Anything
|
||||
else -- non-200 status, non-JSON body, missing/non-int
|
||||
`input_tokens` -- is a regression we want the cell to flip red on.
|
||||
"""
|
||||
url = base_url.rstrip("/") + "/v1/messages/count_tokens"
|
||||
try:
|
||||
response = httpx.post(
|
||||
url,
|
||||
headers={
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
# `anthropic-version` is required by Anthropic's native
|
||||
# API and harmless on every other provider the proxy
|
||||
# routes to. Matches what the Claude Code CLI sends
|
||||
# for its own internal `count_tokens` calls.
|
||||
"anthropic-version": "2023-06-01",
|
||||
},
|
||||
json={"model": model, "messages": [{"role": "user", "content": message}]},
|
||||
timeout=timeout,
|
||||
)
|
||||
except httpx.HTTPError as exc:
|
||||
return ProbeResult(status_code=0, body="", error=f"transport: {exc}")
|
||||
|
||||
body = response.text or ""
|
||||
try:
|
||||
payload = response.json() if body else None
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
payload = None
|
||||
|
||||
return ProbeResult(
|
||||
status_code=response.status_code,
|
||||
body=body,
|
||||
payload=payload,
|
||||
)
|
||||
|
||||
|
||||
def probe_tool_search(
|
||||
*,
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
model: str,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
) -> ProbeResult:
|
||||
"""POST to `{base_url}/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool definition and return the result.
|
||||
|
||||
The shape of the tools array is the one Claude Code emits when its
|
||||
MCP-tool-search beta is active: a `tool_search_tool_regex_20251119`
|
||||
discovery tool (name `tool_search_tool_regex`) plus at least one
|
||||
regular user tool to be searched. LiteLLM's
|
||||
`is_tool_search_used` helper keys on the `_20251119`-suffixed type
|
||||
string to decide whether to attach the provider-specific tool-search
|
||||
beta header (`advanced-tool-use-2025-11-20` for Anthropic/Azure,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock). A proxy
|
||||
regression in that translation will surface here as a 400 from
|
||||
the upstream complaining about the tool type or beta header.
|
||||
|
||||
The prompt deliberately does not force a tool call -- the goal is
|
||||
to verify the *request* round-trips without 400 and produces some
|
||||
response, not to test whether the model decided to invoke
|
||||
tool_search. That kind of behavior test would couple this row to
|
||||
Claude Code's model behavior heuristics, which change weekly.
|
||||
"""
|
||||
url = base_url.rstrip("/") + "/v1/messages"
|
||||
payload = {
|
||||
"model": model,
|
||||
"max_tokens": 64,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"If you have a tool to discover other tools, use it to "
|
||||
"find one. Otherwise reply with the word 'done'."
|
||||
),
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
# The tool_search discovery tool itself. Type is the SDK-
|
||||
# version-pinned `_20251119` suffix; name is the canonical
|
||||
# `tool_search_tool_regex` (no suffix) Anthropic accepts.
|
||||
# LiteLLM keys its beta-header translation on the type.
|
||||
{
|
||||
"type": "tool_search_tool_regex_20251119",
|
||||
"name": "tool_search_tool_regex",
|
||||
},
|
||||
# A trivial user tool for the discovery tool to potentially
|
||||
# surface. Without at least one non-search tool the request
|
||||
# is shape-valid but semantically empty; we include one so
|
||||
# the wire shape mirrors what real Claude Code sends.
|
||||
{
|
||||
"name": "add_numbers",
|
||||
"description": "Add two integers",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"a": {"type": "integer"},
|
||||
"b": {"type": "integer"},
|
||||
},
|
||||
"required": ["a", "b"],
|
||||
},
|
||||
},
|
||||
],
|
||||
}
|
||||
try:
|
||||
response = httpx.post(
|
||||
url,
|
||||
headers={
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
"anthropic-version": "2023-06-01",
|
||||
},
|
||||
json=payload,
|
||||
timeout=timeout,
|
||||
)
|
||||
except httpx.HTTPError as exc:
|
||||
return ProbeResult(status_code=0, body="", error=f"transport: {exc}")
|
||||
|
||||
body = response.text or ""
|
||||
try:
|
||||
payload_out = response.json() if body else None
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
payload_out = None
|
||||
|
||||
return ProbeResult(
|
||||
status_code=response.status_code,
|
||||
body=body,
|
||||
payload=payload_out,
|
||||
)
|
||||
|
||||
|
||||
def assert_tool_search_shape(result: ProbeResult) -> Optional[str]:
|
||||
"""Return None on success, else describe the first violation.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. HTTP status is 200 (no 400 from the upstream rejecting the
|
||||
tool_search tool type or a missing beta header).
|
||||
2. Body is valid JSON.
|
||||
3. Body has either `content` (Anthropic-shape passthrough) or
|
||||
`choices` (LiteLLM normalized openai-shape, used by Bedrock
|
||||
Converse). Either is acceptable -- the matrix cares that the
|
||||
proxy *accepts and forwards* tool_search, not that the model
|
||||
actually chose to invoke it. Tool-invocation behavior is a
|
||||
model decision the matrix has no business asserting on.
|
||||
|
||||
The cell goes red when the upstream rejects the tool type, the
|
||||
proxy drops the beta header, or the response shape is unusable.
|
||||
Anything else (model decided to call or not call tool_search) is
|
||||
irrelevant for this row.
|
||||
"""
|
||||
if result.error is not None:
|
||||
return f"transport error: {result.error}"
|
||||
if result.status_code != 200:
|
||||
return f"status {result.status_code}: {result.body[:400]}"
|
||||
if result.payload is None:
|
||||
return f"non-JSON body: {result.body[:400]}"
|
||||
if not isinstance(result.payload, Mapping):
|
||||
return f"body is not a JSON object: {type(result.payload).__name__}"
|
||||
# LiteLLM normalizes some provider responses to OpenAI shape
|
||||
# (`choices`) and passes others through Anthropic-shape (`content`).
|
||||
# Accept either; both prove the proxy round-tripped the request.
|
||||
if "content" not in result.payload and "choices" not in result.payload:
|
||||
return (
|
||||
f"response has neither `content` nor `choices`: "
|
||||
f"keys={sorted(result.payload.keys())}"
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def assert_count_tokens_shape(result: ProbeResult) -> Optional[str]:
|
||||
"""Return None on success, or an error string describing the first violation.
|
||||
|
||||
Acceptance criteria are intentionally minimal:
|
||||
|
||||
1. HTTP status is 200.
|
||||
2. Body is valid JSON.
|
||||
3. Body has an `input_tokens` key whose value is a positive int.
|
||||
|
||||
Anything beyond that (cache token fields, server metadata) is
|
||||
optional and varies by provider/transport. Asserting on extras
|
||||
would create a brittle test that flips red on neutral protocol
|
||||
drift; matrix cells should only go red on functional regressions
|
||||
a Claude Code user would feel.
|
||||
"""
|
||||
if result.error is not None:
|
||||
return f"transport error: {result.error}"
|
||||
if result.status_code != 200:
|
||||
return f"status {result.status_code}: {result.body[:400]}"
|
||||
if result.payload is None:
|
||||
return f"non-JSON body: {result.body[:400]}"
|
||||
if not isinstance(result.payload, Mapping):
|
||||
return f"body is not a JSON object: {type(result.payload).__name__}"
|
||||
tokens = result.payload.get("input_tokens")
|
||||
if not isinstance(tokens, int) or isinstance(tokens, bool):
|
||||
return f"input_tokens missing or not an int: got {tokens!r}"
|
||||
if tokens <= 0:
|
||||
return f"input_tokens must be positive; got {tokens}"
|
||||
return None
|
||||
0
tests/claude_code/long_context_1m/__init__.py
Normal file
0
tests/claude_code/long_context_1m/__init__.py
Normal file
224
tests/claude_code/long_context_1m/test_anthropic.py
Normal file
224
tests/claude_code/long_context_1m/test_anthropic.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""long_context_1m x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI in headless mode with a ~210k-token padded
|
||||
prompt and the `--betas context-1m-2025-08-07` beta header, route
|
||||
through a LiteLLM proxy aimed at Anthropic, and assert the request
|
||||
round-trips: no 400 from a stripped beta header, no 413 from a body
|
||||
the proxy refused to forward, and a non-empty assistant reply at the
|
||||
end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/long_context_1m/test_anthropic.py
|
||||
^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Cost note (read this before scaling the prompt up):
|
||||
|
||||
This row genuinely exercises the long-context path -- a 210k-token
|
||||
prompt is *just over* Claude's standard 200k context window, which is
|
||||
the threshold that requires the `context-1m-2025-08-07` beta header
|
||||
to be honored end-to-end. Anything shorter would only test whether
|
||||
the proxy forwards the beta header byte-for-byte; it would not catch
|
||||
provider-side regressions where the header is forwarded but the
|
||||
upstream silently truncates beyond the standard context (we've seen
|
||||
this on third-party gateways). Anything longer is wasted spend.
|
||||
|
||||
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
|
||||
Daily cost across all five providers (this row only): ~$19.
|
||||
|
||||
Haiku 4.5 is intentionally omitted: it does not support 1M context
|
||||
(its window is 200k). Reporting `not_applicable` for Haiku would
|
||||
flip the entire cell to `not_applicable`, hiding genuine 1M
|
||||
regressions on Sonnet/Opus; instead we exclude Haiku from the model
|
||||
list entirely and let the matrix's per-cell aggregator green the
|
||||
cell on Sonnet + Opus passing. This is the one row where the "all
|
||||
three tiers must pass" rule is relaxed; it's relaxed structurally
|
||||
(via the model list), not semantically (via not_applicable), so the
|
||||
matrix builder stays unmodified.
|
||||
|
||||
The prompt is delivered via subprocess stdin rather than a positional
|
||||
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
|
||||
fits within that comfortably, but stdin is safer (no shell escaping
|
||||
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
|
||||
keeps the driver's `extra_args` slot free for the `--betas` flag.
|
||||
|
||||
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
|
||||
that loops three Claude tiers in a single cell can't accidentally
|
||||
spend more than ~$18 on this cell. The cap is twice the expected
|
||||
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
|
||||
provider's pricing changes and the matrix starts spending more than
|
||||
$10/day on this row.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Sequence
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
|
||||
# 1M-context beta. See module docstring for the per-cell-aggregator
|
||||
# rationale.
|
||||
ANTHROPIC_MODELS: Sequence[str] = (
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
)
|
||||
|
||||
# Beta header that opts an Anthropic-shape model into the 1M context
|
||||
# window. Same string is accepted on Bedrock (Invoke + Converse) and
|
||||
# Vertex per LiteLLM's transformers -- no per-provider translation is
|
||||
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
|
||||
# `tool-search-tool-2025-10-19`.
|
||||
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
|
||||
|
||||
# Target a padded prompt that lands just above Claude's standard 200k
|
||||
# context window so the request can only succeed if the
|
||||
# `context-1m-2025-08-07` beta header survives all the way to the
|
||||
# upstream. Below 200k the cell would silently pass even with a
|
||||
# proxy-dropped beta header; above ~220k we're paying for tokens that
|
||||
# don't add signal.
|
||||
TARGET_INPUT_TOKENS = 210_000
|
||||
|
||||
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
|
||||
# through several benign pangrams + filler so the padding looks like a
|
||||
# real document, not a repeating monolith. Identical-line padding +
|
||||
# "ignore everything above" trips Opus 4.7's safety filter as a
|
||||
# suspected prompt-injection attempt -- we hit that during smoke
|
||||
# testing and the cell flipped red for the wrong reason. Varied prose
|
||||
# with a natural document-style framing keeps the filter quiet.
|
||||
_PAD_CHUNKS = (
|
||||
"The quick brown fox jumps over the lazy dog. ",
|
||||
"She sells seashells by the seashore on Sunday mornings. ",
|
||||
"Pack my box with five dozen liquor jugs for the journey. ",
|
||||
"How vexingly quick daft zebras jump over fences at dawn. ",
|
||||
"Sphinx of black quartz, judge my vow of silence and patience. ",
|
||||
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
|
||||
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
|
||||
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
|
||||
)
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
|
||||
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
|
||||
|
||||
Framing:
|
||||
|
||||
- Lead with a benign document-style preamble that justifies the
|
||||
long context (so safety filters see the prompt as "long
|
||||
document review" rather than "adversarial padding").
|
||||
- Cycle through a small set of pangrams + filler sentences for
|
||||
the bulk of the padding. Variety matters: identical repeated
|
||||
lines look like a denial-of-service or injection attempt to
|
||||
Anthropic's content filter on the larger tiers.
|
||||
- End with the actual question. Claude's instruction-following
|
||||
is stronger on recent tokens, so a 210k-token-into-the-past
|
||||
instruction would risk a false-fail where the model ignores
|
||||
it.
|
||||
|
||||
`target_tokens` is an approximation: actual token count depends
|
||||
on the tokenizer, but Anthropic's English tokenizer averages
|
||||
~4 chars/token, so 4 × target_tokens chars of padding gets us
|
||||
close enough to the 1M-beta threshold (200k) that the proxy's
|
||||
beta-header handling is the only path to success.
|
||||
"""
|
||||
preamble = (
|
||||
"I'm going to share an excerpt from a long document with you. "
|
||||
"It contains a mix of practice sentences a typist might use to "
|
||||
"warm up; treat the bulk of the text as background context. "
|
||||
"I'll ask a short question at the end.\n\n"
|
||||
"Begin excerpt:\n\n"
|
||||
)
|
||||
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
|
||||
|
||||
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
|
||||
pad_lines = []
|
||||
pad_len = 0
|
||||
idx = 0
|
||||
while pad_len < pad_target_chars:
|
||||
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
|
||||
pad_lines.append(chunk)
|
||||
pad_len += len(chunk)
|
||||
idx += 1
|
||||
return preamble + "".join(pad_lines) + closing
|
||||
|
||||
|
||||
def test_long_context_1m_anthropic(compat_result):
|
||||
"""Drive the `claude` CLI with a ~210k-token prompt and the
|
||||
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
|
||||
non-empty reply for Sonnet + Opus."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
long_prompt = _build_long_prompt()
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=ANTHROPIC_MODELS,
|
||||
prompt=None,
|
||||
stdin_input=long_prompt,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=[
|
||||
"--betas",
|
||||
LONG_CONTEXT_BETA,
|
||||
# Hard ceiling so a runaway test cannot blow the budget.
|
||||
# See module docstring for sizing.
|
||||
"--max-budget-usd",
|
||||
"6",
|
||||
],
|
||||
# Long-context requests can take a couple of minutes on a
|
||||
# loaded upstream; the driver's default 120s is too tight.
|
||||
timeout=300.0,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in ANTHROPIC_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not outcome.text.strip():
|
||||
error = f"[{model}] claude returned empty assistant text"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
224
tests/claude_code/long_context_1m/test_azure.py
Normal file
224
tests/claude_code/long_context_1m/test_azure.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""long_context_1m x Azure (Microsoft Foundry).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with a ~210k-token padded
|
||||
prompt and the `--betas context-1m-2025-08-07` beta header, route
|
||||
through a LiteLLM proxy aimed at Anthropic, and assert the request
|
||||
round-trips: no 400 from a stripped beta header, no 413 from a body
|
||||
the proxy refused to forward, and a non-empty assistant reply at the
|
||||
end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/long_context_1m/test_azure.py
|
||||
^^^^^^^^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
|
||||
Cost note (read this before scaling the prompt up):
|
||||
|
||||
This row genuinely exercises the long-context path -- a 210k-token
|
||||
prompt is *just over* Claude's standard 200k context window, which is
|
||||
the threshold that requires the `context-1m-2025-08-07` beta header
|
||||
to be honored end-to-end. Anything shorter would only test whether
|
||||
the proxy forwards the beta header byte-for-byte; it would not catch
|
||||
provider-side regressions where the header is forwarded but the
|
||||
upstream silently truncates beyond the standard context (we've seen
|
||||
this on third-party gateways). Anything longer is wasted spend.
|
||||
|
||||
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
|
||||
Daily cost across all five providers (this row only): ~$19.
|
||||
|
||||
Haiku 4.5 is intentionally omitted: it does not support 1M context
|
||||
(its window is 200k). Reporting `not_applicable` for Haiku would
|
||||
flip the entire cell to `not_applicable`, hiding genuine 1M
|
||||
regressions on Sonnet/Opus; instead we exclude Haiku from the model
|
||||
list entirely and let the matrix's per-cell aggregator green the
|
||||
cell on Sonnet + Opus passing. This is the one row where the "all
|
||||
three tiers must pass" rule is relaxed; it's relaxed structurally
|
||||
(via the model list), not semantically (via not_applicable), so the
|
||||
matrix builder stays unmodified.
|
||||
|
||||
The prompt is delivered via subprocess stdin rather than a positional
|
||||
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
|
||||
fits within that comfortably, but stdin is safer (no shell escaping
|
||||
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
|
||||
keeps the driver's `extra_args` slot free for the `--betas` flag.
|
||||
|
||||
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
|
||||
that loops three Claude tiers in a single cell can't accidentally
|
||||
spend more than ~$18 on this cell. The cap is twice the expected
|
||||
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
|
||||
provider's pricing changes and the matrix starts spending more than
|
||||
$10/day on this row.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Sequence
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
|
||||
# 1M-context beta. See module docstring for the per-cell-aggregator
|
||||
# rationale.
|
||||
AZURE_MODELS: Sequence[str] = (
|
||||
"claude-sonnet-4-6-azure",
|
||||
"claude-opus-4-7-azure",
|
||||
)
|
||||
|
||||
# Beta header that opts an Anthropic-shape model into the 1M context
|
||||
# window. Same string is accepted on Bedrock (Invoke + Converse) and
|
||||
# Vertex per LiteLLM's transformers -- no per-provider translation is
|
||||
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
|
||||
# `tool-search-tool-2025-10-19`.
|
||||
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
|
||||
|
||||
# Target a padded prompt that lands just above Claude's standard 200k
|
||||
# context window so the request can only succeed if the
|
||||
# `context-1m-2025-08-07` beta header survives all the way to the
|
||||
# upstream. Below 200k the cell would silently pass even with a
|
||||
# proxy-dropped beta header; above ~220k we're paying for tokens that
|
||||
# don't add signal.
|
||||
TARGET_INPUT_TOKENS = 210_000
|
||||
|
||||
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
|
||||
# through several benign pangrams + filler so the padding looks like a
|
||||
# real document, not a repeating monolith. Identical-line padding +
|
||||
# "ignore everything above" trips Opus 4.7's safety filter as a
|
||||
# suspected prompt-injection attempt -- we hit that during smoke
|
||||
# testing and the cell flipped red for the wrong reason. Varied prose
|
||||
# with a natural document-style framing keeps the filter quiet.
|
||||
_PAD_CHUNKS = (
|
||||
"The quick brown fox jumps over the lazy dog. ",
|
||||
"She sells seashells by the seashore on Sunday mornings. ",
|
||||
"Pack my box with five dozen liquor jugs for the journey. ",
|
||||
"How vexingly quick daft zebras jump over fences at dawn. ",
|
||||
"Sphinx of black quartz, judge my vow of silence and patience. ",
|
||||
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
|
||||
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
|
||||
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
|
||||
)
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
|
||||
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
|
||||
|
||||
Framing:
|
||||
|
||||
- Lead with a benign document-style preamble that justifies the
|
||||
long context (so safety filters see the prompt as "long
|
||||
document review" rather than "adversarial padding").
|
||||
- Cycle through a small set of pangrams + filler sentences for
|
||||
the bulk of the padding. Variety matters: identical repeated
|
||||
lines look like a denial-of-service or injection attempt to
|
||||
Anthropic's content filter on the larger tiers.
|
||||
- End with the actual question. Claude's instruction-following
|
||||
is stronger on recent tokens, so a 210k-token-into-the-past
|
||||
instruction would risk a false-fail where the model ignores
|
||||
it.
|
||||
|
||||
`target_tokens` is an approximation: actual token count depends
|
||||
on the tokenizer, but Anthropic's English tokenizer averages
|
||||
~4 chars/token, so 4 × target_tokens chars of padding gets us
|
||||
close enough to the 1M-beta threshold (200k) that the proxy's
|
||||
beta-header handling is the only path to success.
|
||||
"""
|
||||
preamble = (
|
||||
"I'm going to share an excerpt from a long document with you. "
|
||||
"It contains a mix of practice sentences a typist might use to "
|
||||
"warm up; treat the bulk of the text as background context. "
|
||||
"I'll ask a short question at the end.\n\n"
|
||||
"Begin excerpt:\n\n"
|
||||
)
|
||||
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
|
||||
|
||||
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
|
||||
pad_lines = []
|
||||
pad_len = 0
|
||||
idx = 0
|
||||
while pad_len < pad_target_chars:
|
||||
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
|
||||
pad_lines.append(chunk)
|
||||
pad_len += len(chunk)
|
||||
idx += 1
|
||||
return preamble + "".join(pad_lines) + closing
|
||||
|
||||
|
||||
def test_long_context_1m_azure(compat_result):
|
||||
"""Drive the `claude` CLI (Azure (Microsoft Foundry)) with a ~210k-token prompt and the
|
||||
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
|
||||
non-empty reply for Sonnet + Opus."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
long_prompt = _build_long_prompt()
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=AZURE_MODELS,
|
||||
prompt=None,
|
||||
stdin_input=long_prompt,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=[
|
||||
"--betas",
|
||||
LONG_CONTEXT_BETA,
|
||||
# Hard ceiling so a runaway test cannot blow the budget.
|
||||
# See module docstring for sizing.
|
||||
"--max-budget-usd",
|
||||
"6",
|
||||
],
|
||||
# Long-context requests can take a couple of minutes on a
|
||||
# loaded upstream; the driver's default 120s is too tight.
|
||||
timeout=300.0,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in AZURE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not outcome.text.strip():
|
||||
error = f"[{model}] claude returned empty assistant text"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
224
tests/claude_code/long_context_1m/test_bedrock_converse.py
Normal file
224
tests/claude_code/long_context_1m/test_bedrock_converse.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""long_context_1m x Bedrock (Converse).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with a ~210k-token padded
|
||||
prompt and the `--betas context-1m-2025-08-07` beta header, route
|
||||
through a LiteLLM proxy aimed at Anthropic, and assert the request
|
||||
round-trips: no 400 from a stripped beta header, no 413 from a body
|
||||
the proxy refused to forward, and a non-empty assistant reply at the
|
||||
end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/long_context_1m/test_bedrock_converse.py
|
||||
^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Cost note (read this before scaling the prompt up):
|
||||
|
||||
This row genuinely exercises the long-context path -- a 210k-token
|
||||
prompt is *just over* Claude's standard 200k context window, which is
|
||||
the threshold that requires the `context-1m-2025-08-07` beta header
|
||||
to be honored end-to-end. Anything shorter would only test whether
|
||||
the proxy forwards the beta header byte-for-byte; it would not catch
|
||||
provider-side regressions where the header is forwarded but the
|
||||
upstream silently truncates beyond the standard context (we've seen
|
||||
this on third-party gateways). Anything longer is wasted spend.
|
||||
|
||||
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
|
||||
Daily cost across all five providers (this row only): ~$19.
|
||||
|
||||
Haiku 4.5 is intentionally omitted: it does not support 1M context
|
||||
(its window is 200k). Reporting `not_applicable` for Haiku would
|
||||
flip the entire cell to `not_applicable`, hiding genuine 1M
|
||||
regressions on Sonnet/Opus; instead we exclude Haiku from the model
|
||||
list entirely and let the matrix's per-cell aggregator green the
|
||||
cell on Sonnet + Opus passing. This is the one row where the "all
|
||||
three tiers must pass" rule is relaxed; it's relaxed structurally
|
||||
(via the model list), not semantically (via not_applicable), so the
|
||||
matrix builder stays unmodified.
|
||||
|
||||
The prompt is delivered via subprocess stdin rather than a positional
|
||||
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
|
||||
fits within that comfortably, but stdin is safer (no shell escaping
|
||||
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
|
||||
keeps the driver's `extra_args` slot free for the `--betas` flag.
|
||||
|
||||
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
|
||||
that loops three Claude tiers in a single cell can't accidentally
|
||||
spend more than ~$18 on this cell. The cap is twice the expected
|
||||
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
|
||||
provider's pricing changes and the matrix starts spending more than
|
||||
$10/day on this row.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Sequence
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
|
||||
# 1M-context beta. See module docstring for the per-cell-aggregator
|
||||
# rationale.
|
||||
BEDROCK_CONVERSE_MODELS: Sequence[str] = (
|
||||
"claude-sonnet-4-6-bedrock-converse",
|
||||
"claude-opus-4-7-bedrock-converse",
|
||||
)
|
||||
|
||||
# Beta header that opts an Anthropic-shape model into the 1M context
|
||||
# window. Same string is accepted on Bedrock (Invoke + Converse) and
|
||||
# Vertex per LiteLLM's transformers -- no per-provider translation is
|
||||
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
|
||||
# `tool-search-tool-2025-10-19`.
|
||||
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
|
||||
|
||||
# Target a padded prompt that lands just above Claude's standard 200k
|
||||
# context window so the request can only succeed if the
|
||||
# `context-1m-2025-08-07` beta header survives all the way to the
|
||||
# upstream. Below 200k the cell would silently pass even with a
|
||||
# proxy-dropped beta header; above ~220k we're paying for tokens that
|
||||
# don't add signal.
|
||||
TARGET_INPUT_TOKENS = 210_000
|
||||
|
||||
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
|
||||
# through several benign pangrams + filler so the padding looks like a
|
||||
# real document, not a repeating monolith. Identical-line padding +
|
||||
# "ignore everything above" trips Opus 4.7's safety filter as a
|
||||
# suspected prompt-injection attempt -- we hit that during smoke
|
||||
# testing and the cell flipped red for the wrong reason. Varied prose
|
||||
# with a natural document-style framing keeps the filter quiet.
|
||||
_PAD_CHUNKS = (
|
||||
"The quick brown fox jumps over the lazy dog. ",
|
||||
"She sells seashells by the seashore on Sunday mornings. ",
|
||||
"Pack my box with five dozen liquor jugs for the journey. ",
|
||||
"How vexingly quick daft zebras jump over fences at dawn. ",
|
||||
"Sphinx of black quartz, judge my vow of silence and patience. ",
|
||||
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
|
||||
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
|
||||
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
|
||||
)
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
|
||||
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
|
||||
|
||||
Framing:
|
||||
|
||||
- Lead with a benign document-style preamble that justifies the
|
||||
long context (so safety filters see the prompt as "long
|
||||
document review" rather than "adversarial padding").
|
||||
- Cycle through a small set of pangrams + filler sentences for
|
||||
the bulk of the padding. Variety matters: identical repeated
|
||||
lines look like a denial-of-service or injection attempt to
|
||||
Anthropic's content filter on the larger tiers.
|
||||
- End with the actual question. Claude's instruction-following
|
||||
is stronger on recent tokens, so a 210k-token-into-the-past
|
||||
instruction would risk a false-fail where the model ignores
|
||||
it.
|
||||
|
||||
`target_tokens` is an approximation: actual token count depends
|
||||
on the tokenizer, but Anthropic's English tokenizer averages
|
||||
~4 chars/token, so 4 × target_tokens chars of padding gets us
|
||||
close enough to the 1M-beta threshold (200k) that the proxy's
|
||||
beta-header handling is the only path to success.
|
||||
"""
|
||||
preamble = (
|
||||
"I'm going to share an excerpt from a long document with you. "
|
||||
"It contains a mix of practice sentences a typist might use to "
|
||||
"warm up; treat the bulk of the text as background context. "
|
||||
"I'll ask a short question at the end.\n\n"
|
||||
"Begin excerpt:\n\n"
|
||||
)
|
||||
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
|
||||
|
||||
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
|
||||
pad_lines = []
|
||||
pad_len = 0
|
||||
idx = 0
|
||||
while pad_len < pad_target_chars:
|
||||
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
|
||||
pad_lines.append(chunk)
|
||||
pad_len += len(chunk)
|
||||
idx += 1
|
||||
return preamble + "".join(pad_lines) + closing
|
||||
|
||||
|
||||
def test_long_context_1m_bedrock_converse(compat_result):
|
||||
"""Drive the `claude` CLI (Bedrock (Converse)) with a ~210k-token prompt and the
|
||||
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
|
||||
non-empty reply for Sonnet + Opus."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
long_prompt = _build_long_prompt()
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_CONVERSE_MODELS,
|
||||
prompt=None,
|
||||
stdin_input=long_prompt,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=[
|
||||
"--betas",
|
||||
LONG_CONTEXT_BETA,
|
||||
# Hard ceiling so a runaway test cannot blow the budget.
|
||||
# See module docstring for sizing.
|
||||
"--max-budget-usd",
|
||||
"6",
|
||||
],
|
||||
# Long-context requests can take a couple of minutes on a
|
||||
# loaded upstream; the driver's default 120s is too tight.
|
||||
timeout=300.0,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_CONVERSE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not outcome.text.strip():
|
||||
error = f"[{model}] claude returned empty assistant text"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
224
tests/claude_code/long_context_1m/test_bedrock_invoke.py
Normal file
224
tests/claude_code/long_context_1m/test_bedrock_invoke.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""long_context_1m x Bedrock (Invoke).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with a ~210k-token padded
|
||||
prompt and the `--betas context-1m-2025-08-07` beta header, route
|
||||
through a LiteLLM proxy aimed at Anthropic, and assert the request
|
||||
round-trips: no 400 from a stripped beta header, no 413 from a body
|
||||
the proxy refused to forward, and a non-empty assistant reply at the
|
||||
end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/long_context_1m/test_bedrock_invoke.py
|
||||
^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Cost note (read this before scaling the prompt up):
|
||||
|
||||
This row genuinely exercises the long-context path -- a 210k-token
|
||||
prompt is *just over* Claude's standard 200k context window, which is
|
||||
the threshold that requires the `context-1m-2025-08-07` beta header
|
||||
to be honored end-to-end. Anything shorter would only test whether
|
||||
the proxy forwards the beta header byte-for-byte; it would not catch
|
||||
provider-side regressions where the header is forwarded but the
|
||||
upstream silently truncates beyond the standard context (we've seen
|
||||
this on third-party gateways). Anything longer is wasted spend.
|
||||
|
||||
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
|
||||
Daily cost across all five providers (this row only): ~$19.
|
||||
|
||||
Haiku 4.5 is intentionally omitted: it does not support 1M context
|
||||
(its window is 200k). Reporting `not_applicable` for Haiku would
|
||||
flip the entire cell to `not_applicable`, hiding genuine 1M
|
||||
regressions on Sonnet/Opus; instead we exclude Haiku from the model
|
||||
list entirely and let the matrix's per-cell aggregator green the
|
||||
cell on Sonnet + Opus passing. This is the one row where the "all
|
||||
three tiers must pass" rule is relaxed; it's relaxed structurally
|
||||
(via the model list), not semantically (via not_applicable), so the
|
||||
matrix builder stays unmodified.
|
||||
|
||||
The prompt is delivered via subprocess stdin rather than a positional
|
||||
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
|
||||
fits within that comfortably, but stdin is safer (no shell escaping
|
||||
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
|
||||
keeps the driver's `extra_args` slot free for the `--betas` flag.
|
||||
|
||||
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
|
||||
that loops three Claude tiers in a single cell can't accidentally
|
||||
spend more than ~$18 on this cell. The cap is twice the expected
|
||||
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
|
||||
provider's pricing changes and the matrix starts spending more than
|
||||
$10/day on this row.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Sequence
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
|
||||
# 1M-context beta. See module docstring for the per-cell-aggregator
|
||||
# rationale.
|
||||
BEDROCK_INVOKE_MODELS: Sequence[str] = (
|
||||
"claude-sonnet-4-6-bedrock-invoke",
|
||||
"claude-opus-4-7-bedrock-invoke",
|
||||
)
|
||||
|
||||
# Beta header that opts an Anthropic-shape model into the 1M context
|
||||
# window. Same string is accepted on Bedrock (Invoke + Converse) and
|
||||
# Vertex per LiteLLM's transformers -- no per-provider translation is
|
||||
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
|
||||
# `tool-search-tool-2025-10-19`.
|
||||
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
|
||||
|
||||
# Target a padded prompt that lands just above Claude's standard 200k
|
||||
# context window so the request can only succeed if the
|
||||
# `context-1m-2025-08-07` beta header survives all the way to the
|
||||
# upstream. Below 200k the cell would silently pass even with a
|
||||
# proxy-dropped beta header; above ~220k we're paying for tokens that
|
||||
# don't add signal.
|
||||
TARGET_INPUT_TOKENS = 210_000
|
||||
|
||||
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
|
||||
# through several benign pangrams + filler so the padding looks like a
|
||||
# real document, not a repeating monolith. Identical-line padding +
|
||||
# "ignore everything above" trips Opus 4.7's safety filter as a
|
||||
# suspected prompt-injection attempt -- we hit that during smoke
|
||||
# testing and the cell flipped red for the wrong reason. Varied prose
|
||||
# with a natural document-style framing keeps the filter quiet.
|
||||
_PAD_CHUNKS = (
|
||||
"The quick brown fox jumps over the lazy dog. ",
|
||||
"She sells seashells by the seashore on Sunday mornings. ",
|
||||
"Pack my box with five dozen liquor jugs for the journey. ",
|
||||
"How vexingly quick daft zebras jump over fences at dawn. ",
|
||||
"Sphinx of black quartz, judge my vow of silence and patience. ",
|
||||
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
|
||||
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
|
||||
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
|
||||
)
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
|
||||
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
|
||||
|
||||
Framing:
|
||||
|
||||
- Lead with a benign document-style preamble that justifies the
|
||||
long context (so safety filters see the prompt as "long
|
||||
document review" rather than "adversarial padding").
|
||||
- Cycle through a small set of pangrams + filler sentences for
|
||||
the bulk of the padding. Variety matters: identical repeated
|
||||
lines look like a denial-of-service or injection attempt to
|
||||
Anthropic's content filter on the larger tiers.
|
||||
- End with the actual question. Claude's instruction-following
|
||||
is stronger on recent tokens, so a 210k-token-into-the-past
|
||||
instruction would risk a false-fail where the model ignores
|
||||
it.
|
||||
|
||||
`target_tokens` is an approximation: actual token count depends
|
||||
on the tokenizer, but Anthropic's English tokenizer averages
|
||||
~4 chars/token, so 4 × target_tokens chars of padding gets us
|
||||
close enough to the 1M-beta threshold (200k) that the proxy's
|
||||
beta-header handling is the only path to success.
|
||||
"""
|
||||
preamble = (
|
||||
"I'm going to share an excerpt from a long document with you. "
|
||||
"It contains a mix of practice sentences a typist might use to "
|
||||
"warm up; treat the bulk of the text as background context. "
|
||||
"I'll ask a short question at the end.\n\n"
|
||||
"Begin excerpt:\n\n"
|
||||
)
|
||||
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
|
||||
|
||||
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
|
||||
pad_lines = []
|
||||
pad_len = 0
|
||||
idx = 0
|
||||
while pad_len < pad_target_chars:
|
||||
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
|
||||
pad_lines.append(chunk)
|
||||
pad_len += len(chunk)
|
||||
idx += 1
|
||||
return preamble + "".join(pad_lines) + closing
|
||||
|
||||
|
||||
def test_long_context_1m_bedrock_invoke(compat_result):
|
||||
"""Drive the `claude` CLI (Bedrock (Invoke)) with a ~210k-token prompt and the
|
||||
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
|
||||
non-empty reply for Sonnet + Opus."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
long_prompt = _build_long_prompt()
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_INVOKE_MODELS,
|
||||
prompt=None,
|
||||
stdin_input=long_prompt,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=[
|
||||
"--betas",
|
||||
LONG_CONTEXT_BETA,
|
||||
# Hard ceiling so a runaway test cannot blow the budget.
|
||||
# See module docstring for sizing.
|
||||
"--max-budget-usd",
|
||||
"6",
|
||||
],
|
||||
# Long-context requests can take a couple of minutes on a
|
||||
# loaded upstream; the driver's default 120s is too tight.
|
||||
timeout=300.0,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_INVOKE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not outcome.text.strip():
|
||||
error = f"[{model}] claude returned empty assistant text"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
224
tests/claude_code/long_context_1m/test_vertex_ai.py
Normal file
224
tests/claude_code/long_context_1m/test_vertex_ai.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""long_context_1m x Vertex AI.
|
||||
|
||||
Drive the real `claude` CLI in headless mode with a ~210k-token padded
|
||||
prompt and the `--betas context-1m-2025-08-07` beta header, route
|
||||
through a LiteLLM proxy aimed at Anthropic, and assert the request
|
||||
round-trips: no 400 from a stripped beta header, no 413 from a body
|
||||
the proxy refused to forward, and a non-empty assistant reply at the
|
||||
end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/long_context_1m/test_vertex_ai.py
|
||||
^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Cost note (read this before scaling the prompt up):
|
||||
|
||||
This row genuinely exercises the long-context path -- a 210k-token
|
||||
prompt is *just over* Claude's standard 200k context window, which is
|
||||
the threshold that requires the `context-1m-2025-08-07` beta header
|
||||
to be honored end-to-end. Anything shorter would only test whether
|
||||
the proxy forwards the beta header byte-for-byte; it would not catch
|
||||
provider-side regressions where the header is forwarded but the
|
||||
upstream silently truncates beyond the standard context (we've seen
|
||||
this on third-party gateways). Anything longer is wasted spend.
|
||||
|
||||
Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus.
|
||||
Daily cost across all five providers (this row only): ~$19.
|
||||
|
||||
Haiku 4.5 is intentionally omitted: it does not support 1M context
|
||||
(its window is 200k). Reporting `not_applicable` for Haiku would
|
||||
flip the entire cell to `not_applicable`, hiding genuine 1M
|
||||
regressions on Sonnet/Opus; instead we exclude Haiku from the model
|
||||
list entirely and let the matrix's per-cell aggregator green the
|
||||
cell on Sonnet + Opus passing. This is the one row where the "all
|
||||
three tiers must pass" rule is relaxed; it's relaxed structurally
|
||||
(via the model list), not semantically (via not_applicable), so the
|
||||
matrix builder stays unmodified.
|
||||
|
||||
The prompt is delivered via subprocess stdin rather than a positional
|
||||
argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt
|
||||
fits within that comfortably, but stdin is safer (no shell escaping
|
||||
surprises, no surprise ARG_MAX clamp on a tightened sandbox) and
|
||||
keeps the driver's `extra_args` slot free for the `--betas` flag.
|
||||
|
||||
`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test
|
||||
that loops three Claude tiers in a single cell can't accidentally
|
||||
spend more than ~$18 on this cell. The cap is twice the expected
|
||||
worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a
|
||||
provider's pricing changes and the matrix starts spending more than
|
||||
$10/day on this row.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import Sequence
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the
|
||||
# 1M-context beta. See module docstring for the per-cell-aggregator
|
||||
# rationale.
|
||||
VERTEX_AI_MODELS: Sequence[str] = (
|
||||
"claude-sonnet-4-6-vertex",
|
||||
"claude-opus-4-7-vertex",
|
||||
)
|
||||
|
||||
# Beta header that opts an Anthropic-shape model into the 1M context
|
||||
# window. Same string is accepted on Bedrock (Invoke + Converse) and
|
||||
# Vertex per LiteLLM's transformers -- no per-provider translation is
|
||||
# needed for this header, unlike `advanced-tool-use-2025-11-20` /
|
||||
# `tool-search-tool-2025-10-19`.
|
||||
LONG_CONTEXT_BETA = "context-1m-2025-08-07"
|
||||
|
||||
# Target a padded prompt that lands just above Claude's standard 200k
|
||||
# context window so the request can only succeed if the
|
||||
# `context-1m-2025-08-07` beta header survives all the way to the
|
||||
# upstream. Below 200k the cell would silently pass even with a
|
||||
# proxy-dropped beta header; above ~220k we're paying for tokens that
|
||||
# don't add signal.
|
||||
TARGET_INPUT_TOKENS = 210_000
|
||||
|
||||
# Anthropic's English tokenizer averages ~4 chars/token. We cycle
|
||||
# through several benign pangrams + filler so the padding looks like a
|
||||
# real document, not a repeating monolith. Identical-line padding +
|
||||
# "ignore everything above" trips Opus 4.7's safety filter as a
|
||||
# suspected prompt-injection attempt -- we hit that during smoke
|
||||
# testing and the cell flipped red for the wrong reason. Varied prose
|
||||
# with a natural document-style framing keeps the filter quiet.
|
||||
_PAD_CHUNKS = (
|
||||
"The quick brown fox jumps over the lazy dog. ",
|
||||
"She sells seashells by the seashore on Sunday mornings. ",
|
||||
"Pack my box with five dozen liquor jugs for the journey. ",
|
||||
"How vexingly quick daft zebras jump over fences at dawn. ",
|
||||
"Sphinx of black quartz, judge my vow of silence and patience. ",
|
||||
"Waltz, bad nymph, for quick jigs in the moonlit meadow. ",
|
||||
"Glib jocks quiz nymph to vex dwarf with a riddle of stone. ",
|
||||
"Crazy Fredrick bought many very exquisite opal jewels lately. ",
|
||||
)
|
||||
_CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str:
|
||||
"""Build a ~target_tokens-token padded prompt with a trailing instruction.
|
||||
|
||||
Framing:
|
||||
|
||||
- Lead with a benign document-style preamble that justifies the
|
||||
long context (so safety filters see the prompt as "long
|
||||
document review" rather than "adversarial padding").
|
||||
- Cycle through a small set of pangrams + filler sentences for
|
||||
the bulk of the padding. Variety matters: identical repeated
|
||||
lines look like a denial-of-service or injection attempt to
|
||||
Anthropic's content filter on the larger tiers.
|
||||
- End with the actual question. Claude's instruction-following
|
||||
is stronger on recent tokens, so a 210k-token-into-the-past
|
||||
instruction would risk a false-fail where the model ignores
|
||||
it.
|
||||
|
||||
`target_tokens` is an approximation: actual token count depends
|
||||
on the tokenizer, but Anthropic's English tokenizer averages
|
||||
~4 chars/token, so 4 × target_tokens chars of padding gets us
|
||||
close enough to the 1M-beta threshold (200k) that the proxy's
|
||||
beta-header handling is the only path to success.
|
||||
"""
|
||||
preamble = (
|
||||
"I'm going to share an excerpt from a long document with you. "
|
||||
"It contains a mix of practice sentences a typist might use to "
|
||||
"warm up; treat the bulk of the text as background context. "
|
||||
"I'll ask a short question at the end.\n\n"
|
||||
"Begin excerpt:\n\n"
|
||||
)
|
||||
closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'."
|
||||
|
||||
pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing)
|
||||
pad_lines = []
|
||||
pad_len = 0
|
||||
idx = 0
|
||||
while pad_len < pad_target_chars:
|
||||
chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)]
|
||||
pad_lines.append(chunk)
|
||||
pad_len += len(chunk)
|
||||
idx += 1
|
||||
return preamble + "".join(pad_lines) + closing
|
||||
|
||||
|
||||
def test_long_context_1m_vertex_ai(compat_result):
|
||||
"""Drive the `claude` CLI (Vertex AI) with a ~210k-token prompt and the
|
||||
`context-1m-2025-08-07` beta header; assert no 400 / 413 and a
|
||||
non-empty reply for Sonnet + Opus."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
long_prompt = _build_long_prompt()
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=VERTEX_AI_MODELS,
|
||||
prompt=None,
|
||||
stdin_input=long_prompt,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=[
|
||||
"--betas",
|
||||
LONG_CONTEXT_BETA,
|
||||
# Hard ceiling so a runaway test cannot blow the budget.
|
||||
# See module docstring for sizing.
|
||||
"--max-budget-usd",
|
||||
"6",
|
||||
],
|
||||
# Long-context requests can take a couple of minutes on a
|
||||
# loaded upstream; the driver's default 120s is too tight.
|
||||
timeout=300.0,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in VERTEX_AI_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not outcome.text.strip():
|
||||
error = f"[{model}] claude returned empty assistant text"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
|
|
@ -32,8 +32,17 @@ features:
|
|||
name: Prompt caching (5m TTL)
|
||||
- id: vision
|
||||
name: Vision
|
||||
- id: extended_thinking
|
||||
name: Extended thinking
|
||||
- id: thinking
|
||||
name: Thinking
|
||||
# The single row covers both API shapes Anthropic exposes — manual
|
||||
# `thinking: {type: "enabled", budget_tokens: N}` (Haiku 4.5) and
|
||||
# `thinking: {type: "adaptive"}` (Opus 4.7); Sonnet 4.6 supports
|
||||
# either and Claude Code picks per model. A break in either
|
||||
# transformer surfaces as a red cell because all three tiers must
|
||||
# pass for the cell to go green. The row was named
|
||||
# `extended_thinking` historically; Anthropic's docs now reserve
|
||||
# that name for the deprecated manual mode only, so the row was
|
||||
# renamed to the feature-level "Thinking".
|
||||
- id: tool_use_streaming
|
||||
name: Tool use (streaming / fine-grained)
|
||||
- id: thinking_with_tool_use
|
||||
|
|
@ -44,3 +53,51 @@ features:
|
|||
name: Prompt caching (1h TTL)
|
||||
- id: web_search
|
||||
name: Web search (server tool)
|
||||
- id: structured_outputs
|
||||
name: Structured outputs
|
||||
# Drives `claude --json-schema '<schema>'`. Implementation note:
|
||||
# Claude Code translates `--json-schema` to a synthetic
|
||||
# `StructuredOutput` tool whose `input_schema` is the user's
|
||||
# schema, then surfaces the tool_use input as
|
||||
# `structured_output: {...}` on the trailing `result` event.
|
||||
# This row tests that proxy-side handling of that tool round-
|
||||
# trips end-to-end. It does NOT test Anthropic's server-side
|
||||
# `output_config.schema` parameter (a separate feature used
|
||||
# internally by Claude Code for session-title generation) --
|
||||
# `output_config` regressions surface in the HTTP-probe rows.
|
||||
- id: count_tokens
|
||||
name: count_tokens endpoint
|
||||
# HTTP-probe row. Sends a direct POST to
|
||||
# `{proxy}/v1/messages/count_tokens` for each Claude tier and
|
||||
# asserts the response is shaped `{"input_tokens": <positive
|
||||
# int>}`. The CLI uses this endpoint internally but never
|
||||
# surfaces its result in stream-json, so the only way to test
|
||||
# the proxy's handling of it is to hit it directly. LiteLLM has
|
||||
# shipped fixes here (e.g. Claude Code release-notes 2.1.121
|
||||
# "Vertex AI count_tokens returning 400 errors for proxy
|
||||
# gateways"), which is exactly the regression class this row
|
||||
# is meant to catch.
|
||||
- id: tool_search
|
||||
name: Tool search (MCP discovery)
|
||||
# HTTP-probe row. Sends a request whose `tools` array includes
|
||||
# a `tool_search_tool_regex_20251119` discovery tool and asserts
|
||||
# the proxy + upstream accept it. This verifies LiteLLM's
|
||||
# per-provider beta-header translation
|
||||
# (`advanced-tool-use-2025-11-20` for Anthropic/Azure,
|
||||
# `tool-search-tool-2025-10-19` for Vertex/Bedrock) is wired up.
|
||||
# We deliberately don't try to trigger Claude Code's MCP-fan-out
|
||||
# heuristic via `--mcp-config` -- that would couple the row to
|
||||
# an internal behavior threshold that changes between Claude
|
||||
# Code releases. The HTTP probe hits the bug surface LiteLLM
|
||||
# has actually shipped fixes for (2.1.117, 2.1.72, 2.1.70 per
|
||||
# the Claude Code release notes).
|
||||
- id: long_context_1m
|
||||
name: Long context (1M)
|
||||
# Sends a ~210k-token padded prompt with the
|
||||
# `context-1m-2025-08-07` beta header. Just-above the standard
|
||||
# 200k context window so the request can only succeed when the
|
||||
# beta header makes it all the way through the proxy to the
|
||||
# upstream. Haiku 4.5 is intentionally omitted from this row's
|
||||
# model list (its window is 200k); Sonnet 4.6 and Opus 4.7 are
|
||||
# the only tiers exercised. Costs roughly $4/cell/run --
|
||||
# tighten the prompt-token target if pricing changes meaningfully.
|
||||
|
|
|
|||
|
|
@ -78,7 +78,7 @@ PATH="$HOME/.local/bin:$PATH" \
|
|||
uv run pytest \
|
||||
tests/claude_code/basic_messaging_non_streaming \
|
||||
tests/claude_code/basic_messaging_streaming \
|
||||
tests/claude_code/extended_thinking \
|
||||
tests/claude_code/thinking \
|
||||
tests/claude_code/tool_use \
|
||||
tests/claude_code/vision \
|
||||
tests/claude_code/prompt_caching_5m \
|
||||
|
|
|
|||
|
|
@ -117,8 +117,8 @@
|
|||
}
|
||||
},
|
||||
{
|
||||
"id": "extended_thinking",
|
||||
"name": "Extended thinking",
|
||||
"id": "thinking",
|
||||
"name": "Thinking",
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"status": "pass"
|
||||
|
|
|
|||
0
tests/claude_code/structured_outputs/__init__.py
Normal file
0
tests/claude_code/structured_outputs/__init__.py
Normal file
220
tests/claude_code/structured_outputs/test_anthropic.py
Normal file
220
tests/claude_code/structured_outputs/test_anthropic.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
"""structured_outputs x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI in headless mode with the `--json-schema`
|
||||
flag, route through a LiteLLM proxy aimed at Anthropic, and assert that
|
||||
the final stream-json `result` event surfaces a `structured_output`
|
||||
object whose shape matches the schema.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/structured_outputs/test_anthropic.py
|
||||
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
What this row actually exercises (and what it does not):
|
||||
|
||||
`--json-schema` is implemented client-side by Claude Code: the CLI
|
||||
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
|
||||
equals the user-supplied JSON Schema, forces the model toward it, and
|
||||
finally extracts the tool_use input on the trailing `result` event as
|
||||
`structured_output: {...}`. The proxy never sees `output_config.schema`
|
||||
in this flow -- it sees a normal `tools` array with one synthetic
|
||||
tool.
|
||||
|
||||
This makes the row a tool-use feature test in disguise. It's still a
|
||||
distinct row from `tool_use` because:
|
||||
|
||||
- The synthetic tool is generated per request from a user schema, not
|
||||
a developer-declared one. Provider-side bugs that special-case
|
||||
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
|
||||
blocks back to lowercase, or stripping the StructuredOutput-only
|
||||
`additionalProperties: false`) only surface here.
|
||||
- The success signal lives on the *final* `result` event, not the
|
||||
intermediate `assistant` events the `tool_use` row checks. A proxy
|
||||
that drops trailing events (seen in early Bedrock Converse SSE
|
||||
plumbing) breaks this cell while leaving `tool_use` green.
|
||||
|
||||
It is NOT a test of Anthropic's server-side `output_config.schema`
|
||||
parameter -- that's a different feature used internally by Claude Code
|
||||
for session-title generation and is not reachable from any CLI flag.
|
||||
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
|
||||
the `count_tokens` and other HTTP-probe rows, not here.
|
||||
|
||||
Three Claude tiers run in parallel; one `compat_result.add(...)` per
|
||||
tier so the matrix's "all three must pass" rule applies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
ANTHROPIC_MODELS = [
|
||||
"claude-haiku-4-5",
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
# Minimal schema with one required integer field. Kept intentionally
|
||||
# small -- the matrix tests the *plumbing*, not the model's ability to
|
||||
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
|
||||
# integer schema gives every tier (including Haiku) enough headroom
|
||||
# that schema satisfaction is essentially deterministic, isolating
|
||||
# failures to the proxy / transport.
|
||||
SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"answer": {"type": "integer"}},
|
||||
"required": ["answer"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
|
||||
|
||||
# A prompt the model has no reason to misanswer; we don't check the
|
||||
# value, but a wrong answer would suggest the structured-output
|
||||
# pathway is silently degrading reasoning, which is itself worth
|
||||
# noticing.
|
||||
PROMPT = "What is 2 + 2? Reply only via the structured output."
|
||||
|
||||
|
||||
def _extract_structured_output(
|
||||
events: Sequence[Mapping[str, Any]],
|
||||
) -> Optional[Mapping[str, Any]]:
|
||||
"""Return the `structured_output` payload from the last `result` event.
|
||||
|
||||
Claude Code emits its terminal stream-json line as
|
||||
`{"type":"result","structured_output":{...},...}` when a request
|
||||
used `--json-schema` and the model actually produced a valid tool
|
||||
call. If the model bailed or the proxy ate the trailing events,
|
||||
`structured_output` is missing -- which is exactly the failure
|
||||
mode we want this row to surface, so the caller treats `None` as
|
||||
"feature did not work end-to-end".
|
||||
"""
|
||||
for event in reversed(list(events)):
|
||||
if event.get("type") != "result":
|
||||
continue
|
||||
so = event.get("structured_output")
|
||||
if isinstance(so, Mapping):
|
||||
return so
|
||||
return None
|
||||
|
||||
|
||||
def _validate_against_schema(
|
||||
payload: Mapping[str, Any], schema: Mapping[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""Tiny shape validator covering the subset we actually need.
|
||||
|
||||
We deliberately do not pull in `jsonschema` as a test dep: the
|
||||
matrix's success signal is "does the proxy let the synthetic
|
||||
StructuredOutput tool round-trip end-to-end", and that's
|
||||
answerable with a presence + type check over `required` keys.
|
||||
Any malformed schema beyond that would be a Claude Code bug,
|
||||
not a LiteLLM-proxy bug, so a deeper check would only add false
|
||||
failures on the wrong axis.
|
||||
"""
|
||||
type_map = {
|
||||
"integer": int,
|
||||
"number": (int, float),
|
||||
"string": str,
|
||||
"boolean": bool,
|
||||
"array": list,
|
||||
"object": Mapping,
|
||||
}
|
||||
required = schema.get("required") or []
|
||||
properties = schema.get("properties") or {}
|
||||
for key in required:
|
||||
if key not in payload:
|
||||
return f"missing required key {key!r}"
|
||||
expected = (properties.get(key) or {}).get("type")
|
||||
if expected and expected in type_map:
|
||||
if not isinstance(payload[key], type_map[expected]):
|
||||
return (
|
||||
f"key {key!r} has wrong type: "
|
||||
f"expected {expected}, got {type(payload[key]).__name__}"
|
||||
)
|
||||
# bool is a subclass of int in Python; reject `True`/`False`
|
||||
# when the schema asked for an integer/number.
|
||||
if expected in ("integer", "number") and isinstance(payload[key], bool):
|
||||
return f"key {key!r} is a bool but schema asked for {expected}"
|
||||
return None
|
||||
|
||||
|
||||
def test_structured_outputs_anthropic(compat_result):
|
||||
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
|
||||
assert the trailing `result` event contains a schema-conforming
|
||||
`structured_output`."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=ANTHROPIC_MODELS,
|
||||
prompt=PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--json-schema", SCHEMA_JSON],
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in ANTHROPIC_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
payload = _extract_structured_output(outcome.events)
|
||||
if payload is None:
|
||||
error = (
|
||||
f"[{model}] no `structured_output` in trailing result event; "
|
||||
"Claude Code's StructuredOutput tool round-trip did not "
|
||||
"complete end-to-end through the proxy"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
shape_error = _validate_against_schema(payload, SCHEMA)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
220
tests/claude_code/structured_outputs/test_azure.py
Normal file
220
tests/claude_code/structured_outputs/test_azure.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
"""structured_outputs x Azure (Microsoft Foundry).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with the `--json-schema`
|
||||
flag, route through a LiteLLM proxy aimed at Azure (Microsoft Foundry), and assert that
|
||||
the final stream-json `result` event surfaces a `structured_output`
|
||||
object whose shape matches the schema.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/structured_outputs/test_azure.py
|
||||
^^^^^^^^^^^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
|
||||
What this row actually exercises (and what it does not):
|
||||
|
||||
`--json-schema` is implemented client-side by Claude Code: the CLI
|
||||
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
|
||||
equals the user-supplied JSON Schema, forces the model toward it, and
|
||||
finally extracts the tool_use input on the trailing `result` event as
|
||||
`structured_output: {...}`. The proxy never sees `output_config.schema`
|
||||
in this flow -- it sees a normal `tools` array with one synthetic
|
||||
tool.
|
||||
|
||||
This makes the row a tool-use feature test in disguise. It's still a
|
||||
distinct row from `tool_use` because:
|
||||
|
||||
- The synthetic tool is generated per request from a user schema, not
|
||||
a developer-declared one. Provider-side bugs that special-case
|
||||
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
|
||||
blocks back to lowercase, or stripping the StructuredOutput-only
|
||||
`additionalProperties: false`) only surface here.
|
||||
- The success signal lives on the *final* `result` event, not the
|
||||
intermediate `assistant` events the `tool_use` row checks. A proxy
|
||||
that drops trailing events (seen in early Bedrock Converse SSE
|
||||
plumbing) breaks this cell while leaving `tool_use` green.
|
||||
|
||||
It is NOT a test of Anthropic's server-side `output_config.schema`
|
||||
parameter -- that's a different feature used internally by Claude Code
|
||||
for session-title generation and is not reachable from any CLI flag.
|
||||
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
|
||||
the `count_tokens` and other HTTP-probe rows, not here.
|
||||
|
||||
Three Claude tiers run in parallel; one `compat_result.add(...)` per
|
||||
tier so the matrix's "all three must pass" rule applies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
AZURE_MODELS = [
|
||||
"claude-haiku-4-5-azure",
|
||||
"claude-sonnet-4-6-azure",
|
||||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
# Minimal schema with one required integer field. Kept intentionally
|
||||
# small -- the matrix tests the *plumbing*, not the model's ability to
|
||||
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
|
||||
# integer schema gives every tier (including Haiku) enough headroom
|
||||
# that schema satisfaction is essentially deterministic, isolating
|
||||
# failures to the proxy / transport.
|
||||
SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"answer": {"type": "integer"}},
|
||||
"required": ["answer"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
|
||||
|
||||
# A prompt the model has no reason to misanswer; we don't check the
|
||||
# value, but a wrong answer would suggest the structured-output
|
||||
# pathway is silently degrading reasoning, which is itself worth
|
||||
# noticing.
|
||||
PROMPT = "What is 2 + 2? Reply only via the structured output."
|
||||
|
||||
|
||||
def _extract_structured_output(
|
||||
events: Sequence[Mapping[str, Any]],
|
||||
) -> Optional[Mapping[str, Any]]:
|
||||
"""Return the `structured_output` payload from the last `result` event.
|
||||
|
||||
Claude Code emits its terminal stream-json line as
|
||||
`{"type":"result","structured_output":{...},...}` when a request
|
||||
used `--json-schema` and the model actually produced a valid tool
|
||||
call. If the model bailed or the proxy ate the trailing events,
|
||||
`structured_output` is missing -- which is exactly the failure
|
||||
mode we want this row to surface, so the caller treats `None` as
|
||||
"feature did not work end-to-end".
|
||||
"""
|
||||
for event in reversed(list(events)):
|
||||
if event.get("type") != "result":
|
||||
continue
|
||||
so = event.get("structured_output")
|
||||
if isinstance(so, Mapping):
|
||||
return so
|
||||
return None
|
||||
|
||||
|
||||
def _validate_against_schema(
|
||||
payload: Mapping[str, Any], schema: Mapping[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""Tiny shape validator covering the subset we actually need.
|
||||
|
||||
We deliberately do not pull in `jsonschema` as a test dep: the
|
||||
matrix's success signal is "does the proxy let the synthetic
|
||||
StructuredOutput tool round-trip end-to-end", and that's
|
||||
answerable with a presence + type check over `required` keys.
|
||||
Any malformed schema beyond that would be a Claude Code bug,
|
||||
not a LiteLLM-proxy bug, so a deeper check would only add false
|
||||
failures on the wrong axis.
|
||||
"""
|
||||
type_map = {
|
||||
"integer": int,
|
||||
"number": (int, float),
|
||||
"string": str,
|
||||
"boolean": bool,
|
||||
"array": list,
|
||||
"object": Mapping,
|
||||
}
|
||||
required = schema.get("required") or []
|
||||
properties = schema.get("properties") or {}
|
||||
for key in required:
|
||||
if key not in payload:
|
||||
return f"missing required key {key!r}"
|
||||
expected = (properties.get(key) or {}).get("type")
|
||||
if expected and expected in type_map:
|
||||
if not isinstance(payload[key], type_map[expected]):
|
||||
return (
|
||||
f"key {key!r} has wrong type: "
|
||||
f"expected {expected}, got {type(payload[key]).__name__}"
|
||||
)
|
||||
# bool is a subclass of int in Python; reject `True`/`False`
|
||||
# when the schema asked for an integer/number.
|
||||
if expected in ("integer", "number") and isinstance(payload[key], bool):
|
||||
return f"key {key!r} is a bool but schema asked for {expected}"
|
||||
return None
|
||||
|
||||
|
||||
def test_structured_outputs_azure(compat_result):
|
||||
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
|
||||
assert the trailing `result` event contains a schema-conforming
|
||||
`structured_output`."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=AZURE_MODELS,
|
||||
prompt=PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--json-schema", SCHEMA_JSON],
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in AZURE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
payload = _extract_structured_output(outcome.events)
|
||||
if payload is None:
|
||||
error = (
|
||||
f"[{model}] no `structured_output` in trailing result event; "
|
||||
"Claude Code's StructuredOutput tool round-trip did not "
|
||||
"complete end-to-end through the proxy"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
shape_error = _validate_against_schema(payload, SCHEMA)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
220
tests/claude_code/structured_outputs/test_bedrock_converse.py
Normal file
220
tests/claude_code/structured_outputs/test_bedrock_converse.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
"""structured_outputs x Bedrock (Converse).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with the `--json-schema`
|
||||
flag, route through a LiteLLM proxy aimed at Bedrock (Converse), and assert that
|
||||
the final stream-json `result` event surfaces a `structured_output`
|
||||
object whose shape matches the schema.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/structured_outputs/test_bedrock_converse.py
|
||||
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
What this row actually exercises (and what it does not):
|
||||
|
||||
`--json-schema` is implemented client-side by Claude Code: the CLI
|
||||
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
|
||||
equals the user-supplied JSON Schema, forces the model toward it, and
|
||||
finally extracts the tool_use input on the trailing `result` event as
|
||||
`structured_output: {...}`. The proxy never sees `output_config.schema`
|
||||
in this flow -- it sees a normal `tools` array with one synthetic
|
||||
tool.
|
||||
|
||||
This makes the row a tool-use feature test in disguise. It's still a
|
||||
distinct row from `tool_use` because:
|
||||
|
||||
- The synthetic tool is generated per request from a user schema, not
|
||||
a developer-declared one. Provider-side bugs that special-case
|
||||
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
|
||||
blocks back to lowercase, or stripping the StructuredOutput-only
|
||||
`additionalProperties: false`) only surface here.
|
||||
- The success signal lives on the *final* `result` event, not the
|
||||
intermediate `assistant` events the `tool_use` row checks. A proxy
|
||||
that drops trailing events (seen in early Bedrock Converse SSE
|
||||
plumbing) breaks this cell while leaving `tool_use` green.
|
||||
|
||||
It is NOT a test of Anthropic's server-side `output_config.schema`
|
||||
parameter -- that's a different feature used internally by Claude Code
|
||||
for session-title generation and is not reachable from any CLI flag.
|
||||
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
|
||||
the `count_tokens` and other HTTP-probe rows, not here.
|
||||
|
||||
Three Claude tiers run in parallel; one `compat_result.add(...)` per
|
||||
tier so the matrix's "all three must pass" rule applies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_CONVERSE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-converse",
|
||||
"claude-sonnet-4-6-bedrock-converse",
|
||||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
# Minimal schema with one required integer field. Kept intentionally
|
||||
# small -- the matrix tests the *plumbing*, not the model's ability to
|
||||
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
|
||||
# integer schema gives every tier (including Haiku) enough headroom
|
||||
# that schema satisfaction is essentially deterministic, isolating
|
||||
# failures to the proxy / transport.
|
||||
SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"answer": {"type": "integer"}},
|
||||
"required": ["answer"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
|
||||
|
||||
# A prompt the model has no reason to misanswer; we don't check the
|
||||
# value, but a wrong answer would suggest the structured-output
|
||||
# pathway is silently degrading reasoning, which is itself worth
|
||||
# noticing.
|
||||
PROMPT = "What is 2 + 2? Reply only via the structured output."
|
||||
|
||||
|
||||
def _extract_structured_output(
|
||||
events: Sequence[Mapping[str, Any]],
|
||||
) -> Optional[Mapping[str, Any]]:
|
||||
"""Return the `structured_output` payload from the last `result` event.
|
||||
|
||||
Claude Code emits its terminal stream-json line as
|
||||
`{"type":"result","structured_output":{...},...}` when a request
|
||||
used `--json-schema` and the model actually produced a valid tool
|
||||
call. If the model bailed or the proxy ate the trailing events,
|
||||
`structured_output` is missing -- which is exactly the failure
|
||||
mode we want this row to surface, so the caller treats `None` as
|
||||
"feature did not work end-to-end".
|
||||
"""
|
||||
for event in reversed(list(events)):
|
||||
if event.get("type") != "result":
|
||||
continue
|
||||
so = event.get("structured_output")
|
||||
if isinstance(so, Mapping):
|
||||
return so
|
||||
return None
|
||||
|
||||
|
||||
def _validate_against_schema(
|
||||
payload: Mapping[str, Any], schema: Mapping[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""Tiny shape validator covering the subset we actually need.
|
||||
|
||||
We deliberately do not pull in `jsonschema` as a test dep: the
|
||||
matrix's success signal is "does the proxy let the synthetic
|
||||
StructuredOutput tool round-trip end-to-end", and that's
|
||||
answerable with a presence + type check over `required` keys.
|
||||
Any malformed schema beyond that would be a Claude Code bug,
|
||||
not a LiteLLM-proxy bug, so a deeper check would only add false
|
||||
failures on the wrong axis.
|
||||
"""
|
||||
type_map = {
|
||||
"integer": int,
|
||||
"number": (int, float),
|
||||
"string": str,
|
||||
"boolean": bool,
|
||||
"array": list,
|
||||
"object": Mapping,
|
||||
}
|
||||
required = schema.get("required") or []
|
||||
properties = schema.get("properties") or {}
|
||||
for key in required:
|
||||
if key not in payload:
|
||||
return f"missing required key {key!r}"
|
||||
expected = (properties.get(key) or {}).get("type")
|
||||
if expected and expected in type_map:
|
||||
if not isinstance(payload[key], type_map[expected]):
|
||||
return (
|
||||
f"key {key!r} has wrong type: "
|
||||
f"expected {expected}, got {type(payload[key]).__name__}"
|
||||
)
|
||||
# bool is a subclass of int in Python; reject `True`/`False`
|
||||
# when the schema asked for an integer/number.
|
||||
if expected in ("integer", "number") and isinstance(payload[key], bool):
|
||||
return f"key {key!r} is a bool but schema asked for {expected}"
|
||||
return None
|
||||
|
||||
|
||||
def test_structured_outputs_bedrock_converse(compat_result):
|
||||
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
|
||||
assert the trailing `result` event contains a schema-conforming
|
||||
`structured_output`."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_CONVERSE_MODELS,
|
||||
prompt=PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--json-schema", SCHEMA_JSON],
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_CONVERSE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
payload = _extract_structured_output(outcome.events)
|
||||
if payload is None:
|
||||
error = (
|
||||
f"[{model}] no `structured_output` in trailing result event; "
|
||||
"Claude Code's StructuredOutput tool round-trip did not "
|
||||
"complete end-to-end through the proxy"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
shape_error = _validate_against_schema(payload, SCHEMA)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
220
tests/claude_code/structured_outputs/test_bedrock_invoke.py
Normal file
220
tests/claude_code/structured_outputs/test_bedrock_invoke.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
"""structured_outputs x Bedrock (Invoke).
|
||||
|
||||
Drive the real `claude` CLI in headless mode with the `--json-schema`
|
||||
flag, route through a LiteLLM proxy aimed at Bedrock (Invoke), and assert that
|
||||
the final stream-json `result` event surfaces a `structured_output`
|
||||
object whose shape matches the schema.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/structured_outputs/test_bedrock_invoke.py
|
||||
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
What this row actually exercises (and what it does not):
|
||||
|
||||
`--json-schema` is implemented client-side by Claude Code: the CLI
|
||||
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
|
||||
equals the user-supplied JSON Schema, forces the model toward it, and
|
||||
finally extracts the tool_use input on the trailing `result` event as
|
||||
`structured_output: {...}`. The proxy never sees `output_config.schema`
|
||||
in this flow -- it sees a normal `tools` array with one synthetic
|
||||
tool.
|
||||
|
||||
This makes the row a tool-use feature test in disguise. It's still a
|
||||
distinct row from `tool_use` because:
|
||||
|
||||
- The synthetic tool is generated per request from a user schema, not
|
||||
a developer-declared one. Provider-side bugs that special-case
|
||||
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
|
||||
blocks back to lowercase, or stripping the StructuredOutput-only
|
||||
`additionalProperties: false`) only surface here.
|
||||
- The success signal lives on the *final* `result` event, not the
|
||||
intermediate `assistant` events the `tool_use` row checks. A proxy
|
||||
that drops trailing events (seen in early Bedrock Converse SSE
|
||||
plumbing) breaks this cell while leaving `tool_use` green.
|
||||
|
||||
It is NOT a test of Anthropic's server-side `output_config.schema`
|
||||
parameter -- that's a different feature used internally by Claude Code
|
||||
for session-title generation and is not reachable from any CLI flag.
|
||||
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
|
||||
the `count_tokens` and other HTTP-probe rows, not here.
|
||||
|
||||
Three Claude tiers run in parallel; one `compat_result.add(...)` per
|
||||
tier so the matrix's "all three must pass" rule applies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_INVOKE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-invoke",
|
||||
"claude-sonnet-4-6-bedrock-invoke",
|
||||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
# Minimal schema with one required integer field. Kept intentionally
|
||||
# small -- the matrix tests the *plumbing*, not the model's ability to
|
||||
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
|
||||
# integer schema gives every tier (including Haiku) enough headroom
|
||||
# that schema satisfaction is essentially deterministic, isolating
|
||||
# failures to the proxy / transport.
|
||||
SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"answer": {"type": "integer"}},
|
||||
"required": ["answer"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
|
||||
|
||||
# A prompt the model has no reason to misanswer; we don't check the
|
||||
# value, but a wrong answer would suggest the structured-output
|
||||
# pathway is silently degrading reasoning, which is itself worth
|
||||
# noticing.
|
||||
PROMPT = "What is 2 + 2? Reply only via the structured output."
|
||||
|
||||
|
||||
def _extract_structured_output(
|
||||
events: Sequence[Mapping[str, Any]],
|
||||
) -> Optional[Mapping[str, Any]]:
|
||||
"""Return the `structured_output` payload from the last `result` event.
|
||||
|
||||
Claude Code emits its terminal stream-json line as
|
||||
`{"type":"result","structured_output":{...},...}` when a request
|
||||
used `--json-schema` and the model actually produced a valid tool
|
||||
call. If the model bailed or the proxy ate the trailing events,
|
||||
`structured_output` is missing -- which is exactly the failure
|
||||
mode we want this row to surface, so the caller treats `None` as
|
||||
"feature did not work end-to-end".
|
||||
"""
|
||||
for event in reversed(list(events)):
|
||||
if event.get("type") != "result":
|
||||
continue
|
||||
so = event.get("structured_output")
|
||||
if isinstance(so, Mapping):
|
||||
return so
|
||||
return None
|
||||
|
||||
|
||||
def _validate_against_schema(
|
||||
payload: Mapping[str, Any], schema: Mapping[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""Tiny shape validator covering the subset we actually need.
|
||||
|
||||
We deliberately do not pull in `jsonschema` as a test dep: the
|
||||
matrix's success signal is "does the proxy let the synthetic
|
||||
StructuredOutput tool round-trip end-to-end", and that's
|
||||
answerable with a presence + type check over `required` keys.
|
||||
Any malformed schema beyond that would be a Claude Code bug,
|
||||
not a LiteLLM-proxy bug, so a deeper check would only add false
|
||||
failures on the wrong axis.
|
||||
"""
|
||||
type_map = {
|
||||
"integer": int,
|
||||
"number": (int, float),
|
||||
"string": str,
|
||||
"boolean": bool,
|
||||
"array": list,
|
||||
"object": Mapping,
|
||||
}
|
||||
required = schema.get("required") or []
|
||||
properties = schema.get("properties") or {}
|
||||
for key in required:
|
||||
if key not in payload:
|
||||
return f"missing required key {key!r}"
|
||||
expected = (properties.get(key) or {}).get("type")
|
||||
if expected and expected in type_map:
|
||||
if not isinstance(payload[key], type_map[expected]):
|
||||
return (
|
||||
f"key {key!r} has wrong type: "
|
||||
f"expected {expected}, got {type(payload[key]).__name__}"
|
||||
)
|
||||
# bool is a subclass of int in Python; reject `True`/`False`
|
||||
# when the schema asked for an integer/number.
|
||||
if expected in ("integer", "number") and isinstance(payload[key], bool):
|
||||
return f"key {key!r} is a bool but schema asked for {expected}"
|
||||
return None
|
||||
|
||||
|
||||
def test_structured_outputs_bedrock_invoke(compat_result):
|
||||
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
|
||||
assert the trailing `result` event contains a schema-conforming
|
||||
`structured_output`."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_INVOKE_MODELS,
|
||||
prompt=PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--json-schema", SCHEMA_JSON],
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_INVOKE_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
payload = _extract_structured_output(outcome.events)
|
||||
if payload is None:
|
||||
error = (
|
||||
f"[{model}] no `structured_output` in trailing result event; "
|
||||
"Claude Code's StructuredOutput tool round-trip did not "
|
||||
"complete end-to-end through the proxy"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
shape_error = _validate_against_schema(payload, SCHEMA)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
220
tests/claude_code/structured_outputs/test_vertex_ai.py
Normal file
220
tests/claude_code/structured_outputs/test_vertex_ai.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
"""structured_outputs x Vertex AI.
|
||||
|
||||
Drive the real `claude` CLI in headless mode with the `--json-schema`
|
||||
flag, route through a LiteLLM proxy aimed at Vertex AI, and assert that
|
||||
the final stream-json `result` event surfaces a `structured_output`
|
||||
object whose shape matches the schema.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/structured_outputs/test_vertex_ai.py
|
||||
^^^^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
What this row actually exercises (and what it does not):
|
||||
|
||||
`--json-schema` is implemented client-side by Claude Code: the CLI
|
||||
synthesizes a synthetic `StructuredOutput` tool whose `input_schema`
|
||||
equals the user-supplied JSON Schema, forces the model toward it, and
|
||||
finally extracts the tool_use input on the trailing `result` event as
|
||||
`structured_output: {...}`. The proxy never sees `output_config.schema`
|
||||
in this flow -- it sees a normal `tools` array with one synthetic
|
||||
tool.
|
||||
|
||||
This makes the row a tool-use feature test in disguise. It's still a
|
||||
distinct row from `tool_use` because:
|
||||
|
||||
- The synthetic tool is generated per request from a user schema, not
|
||||
a developer-declared one. Provider-side bugs that special-case
|
||||
`Claude Code`-generated tool names (e.g. case-folding `tool_use`
|
||||
blocks back to lowercase, or stripping the StructuredOutput-only
|
||||
`additionalProperties: false`) only surface here.
|
||||
- The success signal lives on the *final* `result` event, not the
|
||||
intermediate `assistant` events the `tool_use` row checks. A proxy
|
||||
that drops trailing events (seen in early Bedrock Converse SSE
|
||||
plumbing) breaks this cell while leaving `tool_use` green.
|
||||
|
||||
It is NOT a test of Anthropic's server-side `output_config.schema`
|
||||
parameter -- that's a different feature used internally by Claude Code
|
||||
for session-title generation and is not reachable from any CLI flag.
|
||||
LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in
|
||||
the `count_tokens` and other HTTP-probe rows, not here.
|
||||
|
||||
Three Claude tiers run in parallel; one `compat_result.add(...)` per
|
||||
tier so the matrix's "all three must pass" rule applies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Mapping, Optional, Sequence, Tuple
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.cli_driver import (
|
||||
ClaudeCLIError,
|
||||
failure_diagnostic,
|
||||
run_claude_models_parallel,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
VERTEX_AI_MODELS = [
|
||||
"claude-haiku-4-5-vertex",
|
||||
"claude-sonnet-4-6-vertex",
|
||||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
# Minimal schema with one required integer field. Kept intentionally
|
||||
# small -- the matrix tests the *plumbing*, not the model's ability to
|
||||
# satisfy a complex schema. A trivial arithmetic prompt + a one-field
|
||||
# integer schema gives every tier (including Haiku) enough headroom
|
||||
# that schema satisfaction is essentially deterministic, isolating
|
||||
# failures to the proxy / transport.
|
||||
SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {"answer": {"type": "integer"}},
|
||||
"required": ["answer"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":"))
|
||||
|
||||
# A prompt the model has no reason to misanswer; we don't check the
|
||||
# value, but a wrong answer would suggest the structured-output
|
||||
# pathway is silently degrading reasoning, which is itself worth
|
||||
# noticing.
|
||||
PROMPT = "What is 2 + 2? Reply only via the structured output."
|
||||
|
||||
|
||||
def _extract_structured_output(
|
||||
events: Sequence[Mapping[str, Any]],
|
||||
) -> Optional[Mapping[str, Any]]:
|
||||
"""Return the `structured_output` payload from the last `result` event.
|
||||
|
||||
Claude Code emits its terminal stream-json line as
|
||||
`{"type":"result","structured_output":{...},...}` when a request
|
||||
used `--json-schema` and the model actually produced a valid tool
|
||||
call. If the model bailed or the proxy ate the trailing events,
|
||||
`structured_output` is missing -- which is exactly the failure
|
||||
mode we want this row to surface, so the caller treats `None` as
|
||||
"feature did not work end-to-end".
|
||||
"""
|
||||
for event in reversed(list(events)):
|
||||
if event.get("type") != "result":
|
||||
continue
|
||||
so = event.get("structured_output")
|
||||
if isinstance(so, Mapping):
|
||||
return so
|
||||
return None
|
||||
|
||||
|
||||
def _validate_against_schema(
|
||||
payload: Mapping[str, Any], schema: Mapping[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""Tiny shape validator covering the subset we actually need.
|
||||
|
||||
We deliberately do not pull in `jsonschema` as a test dep: the
|
||||
matrix's success signal is "does the proxy let the synthetic
|
||||
StructuredOutput tool round-trip end-to-end", and that's
|
||||
answerable with a presence + type check over `required` keys.
|
||||
Any malformed schema beyond that would be a Claude Code bug,
|
||||
not a LiteLLM-proxy bug, so a deeper check would only add false
|
||||
failures on the wrong axis.
|
||||
"""
|
||||
type_map = {
|
||||
"integer": int,
|
||||
"number": (int, float),
|
||||
"string": str,
|
||||
"boolean": bool,
|
||||
"array": list,
|
||||
"object": Mapping,
|
||||
}
|
||||
required = schema.get("required") or []
|
||||
properties = schema.get("properties") or {}
|
||||
for key in required:
|
||||
if key not in payload:
|
||||
return f"missing required key {key!r}"
|
||||
expected = (properties.get(key) or {}).get("type")
|
||||
if expected and expected in type_map:
|
||||
if not isinstance(payload[key], type_map[expected]):
|
||||
return (
|
||||
f"key {key!r} has wrong type: "
|
||||
f"expected {expected}, got {type(payload[key]).__name__}"
|
||||
)
|
||||
# bool is a subclass of int in Python; reject `True`/`False`
|
||||
# when the schema asked for an integer/number.
|
||||
if expected in ("integer", "number") and isinstance(payload[key], bool):
|
||||
return f"key {key!r} is a bool but schema asked for {expected}"
|
||||
return None
|
||||
|
||||
|
||||
def test_structured_outputs_vertex_ai(compat_result):
|
||||
"""Drive `claude --json-schema ...` against the LiteLLM proxy and
|
||||
assert the trailing `result` event contains a schema-conforming
|
||||
`structured_output`."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=VERTEX_AI_MODELS,
|
||||
prompt=PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--json-schema", SCHEMA_JSON],
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in VERTEX_AI_MODELS:
|
||||
outcome = outcomes[model]
|
||||
if isinstance(outcome, ClaudeCLIError):
|
||||
error = f"[{model}] {outcome}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
if outcome.exit_code != 0:
|
||||
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
payload = _extract_structured_output(outcome.events)
|
||||
if payload is None:
|
||||
error = (
|
||||
f"[{model}] no `structured_output` in trailing result event; "
|
||||
"Claude Code's StructuredOutput tool round-trip did not "
|
||||
"complete end-to-end through the proxy"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
shape_error = _validate_against_schema(payload, SCHEMA)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
0
tests/claude_code/thinking/__init__.py
Normal file
0
tests/claude_code/thinking/__init__.py
Normal file
|
|
@ -1,4 +1,4 @@
|
|||
"""extended_thinking x Anthropic.
|
||||
"""thinking x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
to Anthropic, enable extended thinking via `--effort high`, and assert
|
||||
|
|
@ -9,9 +9,9 @@ upstream response's `thinking` content blocks end-to-end.
|
|||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/extended_thinking/test_anthropic.py
|
||||
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
tests/claude_code/thinking/test_anthropic.py
|
||||
^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
The three Claude tiers run in parallel inside this single test, with
|
||||
one `compat_result.add(...)` entry per model so the matrix builder
|
||||
|
|
@ -76,7 +76,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def test_extended_thinking_anthropic(compat_result):
|
||||
def test_thinking_anthropic(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
||||
enabled and assert a `thinking` content block was emitted."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
"""extended_thinking x Azure (Microsoft Foundry).
|
||||
"""thinking x Azure (Microsoft Foundry).
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models hosted in Microsoft Foundry on
|
||||
|
|
@ -16,9 +16,9 @@ re-evaluate.
|
|||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/extended_thinking/test_azure.py
|
||||
^^^^^^^^^^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
tests/claude_code/thinking/test_azure.py
|
||||
^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -64,7 +64,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def test_extended_thinking_azure(compat_result):
|
||||
def test_thinking_azure(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
||||
enabled and assert a `thinking` content block was emitted."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
"""extended_thinking x Bedrock (Converse).
|
||||
"""thinking x Bedrock (Converse).
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the unified `Converse` API path,
|
||||
|
|
@ -8,9 +8,9 @@ upstream returned a `thinking` content block.
|
|||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/extended_thinking/test_bedrock_converse.py
|
||||
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
tests/claude_code/thinking/test_bedrock_converse.py
|
||||
^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def test_extended_thinking_bedrock_converse(compat_result):
|
||||
def test_thinking_bedrock_converse(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
||||
enabled and assert a `thinking` content block was emitted."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
"""extended_thinking x Bedrock (Invoke).
|
||||
"""thinking x Bedrock (Invoke).
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
||||
|
|
@ -8,9 +8,9 @@ upstream returned a `thinking` content block.
|
|||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/extended_thinking/test_bedrock_invoke.py
|
||||
^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
tests/claude_code/thinking/test_bedrock_invoke.py
|
||||
^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def test_extended_thinking_bedrock_invoke(compat_result):
|
||||
def test_thinking_bedrock_invoke(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
||||
enabled and assert a `thinking` content block was emitted."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
"""extended_thinking x Vertex AI.
|
||||
"""thinking x Vertex AI.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models on Google Cloud Vertex AI, enable
|
||||
|
|
@ -8,9 +8,9 @@ upstream returned a `thinking` content block.
|
|||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/extended_thinking/test_vertex_ai.py
|
||||
^^^^^^^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
tests/claude_code/thinking/test_vertex_ai.py
|
||||
^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def test_extended_thinking_vertex_ai(compat_result):
|
||||
def test_thinking_vertex_ai(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with thinking
|
||||
enabled and assert a `thinking` content block was emitted."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
0
tests/claude_code/tool_search/__init__.py
Normal file
0
tests/claude_code/tool_search/__init__.py
Normal file
99
tests/claude_code/tool_search/test_anthropic.py
Normal file
99
tests/claude_code/tool_search/test_anthropic.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""tool_search x Anthropic.
|
||||
|
||||
HTTP-probe row. Sends a single `/v1/messages` request whose `tools`
|
||||
array includes a `tool_search_tool_regex_20251119` discovery tool, and
|
||||
asserts the proxy round-trips it to the upstream without a 400. This
|
||||
verifies LiteLLM's tool-search beta-header translation
|
||||
(`advanced-tool-use-2025-11-20` for Anthropic-shape providers,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/tool_search/test_anthropic.py
|
||||
^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI / MCP fan-out:
|
||||
|
||||
Real Claude Code activates tool_search by registering >N MCP tools and
|
||||
relying on the model's internal heuristic to call the discovery tool
|
||||
before any user tool. That setup requires standing up a stub MCP
|
||||
server that exposes 50+ tool stubs and depends on Claude Code's
|
||||
auto-deferral heuristic continuing to fire at today's tool count --
|
||||
both of which break silently when Claude Code's threshold changes
|
||||
between releases.
|
||||
|
||||
The bugs LiteLLM has actually shipped fixes for in this area
|
||||
(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are
|
||||
beta-header translation and proxy-side type recognition, not MCP
|
||||
fan-out behavior. An HTTP probe hits exactly that surface: the
|
||||
request goes out with a `tool_search_tool_regex_20251119` tool type,
|
||||
the proxy is responsible for attaching the per-provider beta header
|
||||
and forwarding, and the upstream either accepts or 400s. A red cell
|
||||
here is always a proxy-side regression, not a flaky model-behavior
|
||||
artifact.
|
||||
|
||||
Three Claude tiers are probed in sequence (count is too low to be
|
||||
worth the parallelism overhead, and HTTP probes don't compete for
|
||||
the proxy's `--num-workers` slots the way CLI subprocess runs do).
|
||||
The matrix's "all three must pass" rule still applies via the
|
||||
per-cell aggregator.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_tool_search_shape,
|
||||
probe_tool_search,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
ANTHROPIC_MODELS = [
|
||||
"claude-haiku-4-5",
|
||||
"claude-sonnet-4-6",
|
||||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
|
||||
def test_tool_search_anthropic(compat_result):
|
||||
"""Probe `/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool and assert the proxy + upstream accept it for every Anthropic
|
||||
tier."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in ANTHROPIC_MODELS:
|
||||
result = probe_tool_search(base_url=base_url, api_key=api_key, model=model)
|
||||
shape_error = assert_tool_search_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] tool_search probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
99
tests/claude_code/tool_search/test_azure.py
Normal file
99
tests/claude_code/tool_search/test_azure.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""tool_search x Azure (Microsoft Foundry).
|
||||
|
||||
HTTP-probe row. Sends a single `/v1/messages` request whose `tools`
|
||||
array includes a `tool_search_tool_regex_20251119` discovery tool, and
|
||||
asserts the proxy round-trips it to the upstream without a 400. This
|
||||
verifies LiteLLM's tool-search beta-header translation
|
||||
(`advanced-tool-use-2025-11-20` for Anthropic-shape providers,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/tool_search/test_azure.py
|
||||
^^^^^^^^^^^ ^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI / MCP fan-out:
|
||||
|
||||
Real Claude Code activates tool_search by registering >N MCP tools and
|
||||
relying on the model's internal heuristic to call the discovery tool
|
||||
before any user tool. That setup requires standing up a stub MCP
|
||||
server that exposes 50+ tool stubs and depends on Claude Code's
|
||||
auto-deferral heuristic continuing to fire at today's tool count --
|
||||
both of which break silently when Claude Code's threshold changes
|
||||
between releases.
|
||||
|
||||
The bugs LiteLLM has actually shipped fixes for in this area
|
||||
(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are
|
||||
beta-header translation and proxy-side type recognition, not MCP
|
||||
fan-out behavior. An HTTP probe hits exactly that surface: the
|
||||
request goes out with a `tool_search_tool_regex_20251119` tool type,
|
||||
the proxy is responsible for attaching the per-provider beta header
|
||||
and forwarding, and the upstream either accepts or 400s. A red cell
|
||||
here is always a proxy-side regression, not a flaky model-behavior
|
||||
artifact.
|
||||
|
||||
Three Claude tiers are probed in sequence (count is too low to be
|
||||
worth the parallelism overhead, and HTTP probes don't compete for
|
||||
the proxy's `--num-workers` slots the way CLI subprocess runs do).
|
||||
The matrix's "all three must pass" rule still applies via the
|
||||
per-cell aggregator.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_tool_search_shape,
|
||||
probe_tool_search,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
AZURE_MODELS = [
|
||||
"claude-haiku-4-5-azure",
|
||||
"claude-sonnet-4-6-azure",
|
||||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
|
||||
def test_tool_search_azure(compat_result):
|
||||
"""Probe `/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool and assert the proxy + upstream accept it for every Azure (Microsoft Foundry)
|
||||
tier."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in AZURE_MODELS:
|
||||
result = probe_tool_search(base_url=base_url, api_key=api_key, model=model)
|
||||
shape_error = assert_tool_search_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] tool_search probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
99
tests/claude_code/tool_search/test_bedrock_converse.py
Normal file
99
tests/claude_code/tool_search/test_bedrock_converse.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""tool_search x Bedrock (Converse).
|
||||
|
||||
HTTP-probe row. Sends a single `/v1/messages` request whose `tools`
|
||||
array includes a `tool_search_tool_regex_20251119` discovery tool, and
|
||||
asserts the proxy round-trips it to the upstream without a 400. This
|
||||
verifies LiteLLM's tool-search beta-header translation
|
||||
(`advanced-tool-use-2025-11-20` for Anthropic-shape providers,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/tool_search/test_bedrock_converse.py
|
||||
^^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI / MCP fan-out:
|
||||
|
||||
Real Claude Code activates tool_search by registering >N MCP tools and
|
||||
relying on the model's internal heuristic to call the discovery tool
|
||||
before any user tool. That setup requires standing up a stub MCP
|
||||
server that exposes 50+ tool stubs and depends on Claude Code's
|
||||
auto-deferral heuristic continuing to fire at today's tool count --
|
||||
both of which break silently when Claude Code's threshold changes
|
||||
between releases.
|
||||
|
||||
The bugs LiteLLM has actually shipped fixes for in this area
|
||||
(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are
|
||||
beta-header translation and proxy-side type recognition, not MCP
|
||||
fan-out behavior. An HTTP probe hits exactly that surface: the
|
||||
request goes out with a `tool_search_tool_regex_20251119` tool type,
|
||||
the proxy is responsible for attaching the per-provider beta header
|
||||
and forwarding, and the upstream either accepts or 400s. A red cell
|
||||
here is always a proxy-side regression, not a flaky model-behavior
|
||||
artifact.
|
||||
|
||||
Three Claude tiers are probed in sequence (count is too low to be
|
||||
worth the parallelism overhead, and HTTP probes don't compete for
|
||||
the proxy's `--num-workers` slots the way CLI subprocess runs do).
|
||||
The matrix's "all three must pass" rule still applies via the
|
||||
per-cell aggregator.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_tool_search_shape,
|
||||
probe_tool_search,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_CONVERSE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-converse",
|
||||
"claude-sonnet-4-6-bedrock-converse",
|
||||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
|
||||
def test_tool_search_bedrock_converse(compat_result):
|
||||
"""Probe `/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool and assert the proxy + upstream accept it for every Bedrock (Converse)
|
||||
tier."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_CONVERSE_MODELS:
|
||||
result = probe_tool_search(base_url=base_url, api_key=api_key, model=model)
|
||||
shape_error = assert_tool_search_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] tool_search probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
99
tests/claude_code/tool_search/test_bedrock_invoke.py
Normal file
99
tests/claude_code/tool_search/test_bedrock_invoke.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""tool_search x Bedrock (Invoke).
|
||||
|
||||
HTTP-probe row. Sends a single `/v1/messages` request whose `tools`
|
||||
array includes a `tool_search_tool_regex_20251119` discovery tool, and
|
||||
asserts the proxy round-trips it to the upstream without a 400. This
|
||||
verifies LiteLLM's tool-search beta-header translation
|
||||
(`advanced-tool-use-2025-11-20` for Anthropic-shape providers,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/tool_search/test_bedrock_invoke.py
|
||||
^^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI / MCP fan-out:
|
||||
|
||||
Real Claude Code activates tool_search by registering >N MCP tools and
|
||||
relying on the model's internal heuristic to call the discovery tool
|
||||
before any user tool. That setup requires standing up a stub MCP
|
||||
server that exposes 50+ tool stubs and depends on Claude Code's
|
||||
auto-deferral heuristic continuing to fire at today's tool count --
|
||||
both of which break silently when Claude Code's threshold changes
|
||||
between releases.
|
||||
|
||||
The bugs LiteLLM has actually shipped fixes for in this area
|
||||
(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are
|
||||
beta-header translation and proxy-side type recognition, not MCP
|
||||
fan-out behavior. An HTTP probe hits exactly that surface: the
|
||||
request goes out with a `tool_search_tool_regex_20251119` tool type,
|
||||
the proxy is responsible for attaching the per-provider beta header
|
||||
and forwarding, and the upstream either accepts or 400s. A red cell
|
||||
here is always a proxy-side regression, not a flaky model-behavior
|
||||
artifact.
|
||||
|
||||
Three Claude tiers are probed in sequence (count is too low to be
|
||||
worth the parallelism overhead, and HTTP probes don't compete for
|
||||
the proxy's `--num-workers` slots the way CLI subprocess runs do).
|
||||
The matrix's "all three must pass" rule still applies via the
|
||||
per-cell aggregator.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_tool_search_shape,
|
||||
probe_tool_search,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
BEDROCK_INVOKE_MODELS = [
|
||||
"claude-haiku-4-5-bedrock-invoke",
|
||||
"claude-sonnet-4-6-bedrock-invoke",
|
||||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
|
||||
def test_tool_search_bedrock_invoke(compat_result):
|
||||
"""Probe `/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool and assert the proxy + upstream accept it for every Bedrock (Invoke)
|
||||
tier."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in BEDROCK_INVOKE_MODELS:
|
||||
result = probe_tool_search(base_url=base_url, api_key=api_key, model=model)
|
||||
shape_error = assert_tool_search_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] tool_search probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
99
tests/claude_code/tool_search/test_vertex_ai.py
Normal file
99
tests/claude_code/tool_search/test_vertex_ai.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""tool_search x Vertex AI.
|
||||
|
||||
HTTP-probe row. Sends a single `/v1/messages` request whose `tools`
|
||||
array includes a `tool_search_tool_regex_20251119` discovery tool, and
|
||||
asserts the proxy round-trips it to the upstream without a 400. This
|
||||
verifies LiteLLM's tool-search beta-header translation
|
||||
(`advanced-tool-use-2025-11-20` for Anthropic-shape providers,
|
||||
`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/tool_search/test_vertex_ai.py
|
||||
^^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why HTTP probe instead of CLI / MCP fan-out:
|
||||
|
||||
Real Claude Code activates tool_search by registering >N MCP tools and
|
||||
relying on the model's internal heuristic to call the discovery tool
|
||||
before any user tool. That setup requires standing up a stub MCP
|
||||
server that exposes 50+ tool stubs and depends on Claude Code's
|
||||
auto-deferral heuristic continuing to fire at today's tool count --
|
||||
both of which break silently when Claude Code's threshold changes
|
||||
between releases.
|
||||
|
||||
The bugs LiteLLM has actually shipped fixes for in this area
|
||||
(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are
|
||||
beta-header translation and proxy-side type recognition, not MCP
|
||||
fan-out behavior. An HTTP probe hits exactly that surface: the
|
||||
request goes out with a `tool_search_tool_regex_20251119` tool type,
|
||||
the proxy is responsible for attaching the per-provider beta header
|
||||
and forwarding, and the upstream either accepts or 400s. A red cell
|
||||
here is always a proxy-side regression, not a flaky model-behavior
|
||||
artifact.
|
||||
|
||||
Three Claude tiers are probed in sequence (count is too low to be
|
||||
worth the parallelism overhead, and HTTP probes don't compete for
|
||||
the proxy's `--num-workers` slots the way CLI subprocess runs do).
|
||||
The matrix's "all three must pass" rule still applies via the
|
||||
per-cell aggregator.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.claude_code.http_probe import (
|
||||
assert_tool_search_shape,
|
||||
probe_tool_search,
|
||||
)
|
||||
|
||||
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
||||
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
||||
|
||||
VERTEX_AI_MODELS = [
|
||||
"claude-haiku-4-5-vertex",
|
||||
"claude-sonnet-4-6-vertex",
|
||||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
|
||||
def test_tool_search_vertex_ai(compat_result):
|
||||
"""Probe `/v1/messages` with a `tool_search_tool_regex_20251119`
|
||||
tool and assert the proxy + upstream accept it for every Vertex AI
|
||||
tier."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
compat_result.set(
|
||||
{
|
||||
"status": "fail",
|
||||
"error": (
|
||||
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
||||
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
||||
),
|
||||
}
|
||||
)
|
||||
pytest.fail(
|
||||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured",
|
||||
pytrace=False,
|
||||
)
|
||||
|
||||
failures = []
|
||||
for model in VERTEX_AI_MODELS:
|
||||
result = probe_tool_search(base_url=base_url, api_key=api_key, model=model)
|
||||
shape_error = assert_tool_search_shape(result)
|
||||
if shape_error is not None:
|
||||
error = f"[{model}] tool_search probe failed: {shape_error}"
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
continue
|
||||
|
||||
compat_result.add({"status": "pass"})
|
||||
|
||||
if failures:
|
||||
pytest.fail("; ".join(failures), pytrace=False)
|
||||
Loading…
Add table
Reference in a new issue