diff --git a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py index 6cf34a05e3f..17411464050 100644 --- a/tests/claude_code/_builder_unit_tests/test_matrix_builder.py +++ b/tests/claude_code/_builder_unit_tests/test_matrix_builder.py @@ -206,7 +206,12 @@ def test_build_matrix_6x5_grid_matches_published_sample(): "tool_use", "prompt_caching_5m", "vision", - "extended_thinking", + # Row 6 of the v0 PRD; originally shipped as `extended_thinking`. + # The id was renamed in-place to `thinking` to match Anthropic's + # current docs (which reserve "extended thinking" for the + # deprecated manual mode only). The row's *position* in v0 is + # the load-bearing invariant, not the id string. + "thinking", ] v0_features = [ feature diff --git a/tests/claude_code/_builder_unit_tests/test_v0_layout.py b/tests/claude_code/_builder_unit_tests/test_v0_layout.py index 19064c15df1..d6533d96da5 100644 --- a/tests/claude_code/_builder_unit_tests/test_v0_layout.py +++ b/tests/claude_code/_builder_unit_tests/test_v0_layout.py @@ -26,7 +26,13 @@ EXPECTED_FEATURE_IDS = [ "tool_use", "prompt_caching_5m", "vision", - "extended_thinking", + # v0 originally shipped this row as `extended_thinking`. It was + # renamed in-place to `thinking` because Anthropic's docs reserve + # "extended thinking" for the deprecated manual API mode only; the + # single row exercises both manual and adaptive shapes since Claude + # Code picks per model. The PRD's "v0" identity is the *position* + # (row 6, 0-indexed 5), not the id string. + "thinking", ] # The PRD's column order. Every feature directory must have one diff --git a/tests/claude_code/extended_thinking/__init__.py b/tests/claude_code/count_tokens/__init__.py similarity index 100% rename from tests/claude_code/extended_thinking/__init__.py rename to tests/claude_code/count_tokens/__init__.py diff --git a/tests/claude_code/count_tokens/test_anthropic.py b/tests/claude_code/count_tokens/test_anthropic.py new file mode 100644 index 00000000000..1ca35bdb860 --- /dev/null +++ b/tests/claude_code/count_tokens/test_anthropic.py @@ -0,0 +1,94 @@ +"""count_tokens x Anthropic. + +HTTP-probe row. Unlike the CLI-driven rows, this test never invokes +the `claude` CLI: it `POST`s directly to +`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts +the response is shaped `{"input_tokens": }`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/count_tokens/test_anthropic.py + ^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI: + +Claude Code calls `count_tokens` internally to compute budget / +context-window usage display, but the result is consumed by the CLI +in-process and never appears in stream-json events. There is no CLI +flag that emits the count to stdout in a way our existing +stream-json parser can pick up, so we can't test the endpoint round +trip through the CLI surface. + +The proxy *is* expected to expose `/v1/messages/count_tokens` for +every Claude-style provider it routes to -- LiteLLM has historically +had provider-specific bugs in this endpoint (Vertex AI `count_tokens` +returned 400 to proxy gateways; see Claude Code release notes 2.1.121). +Treating it as a matrix row keeps regressions in the cron's daily +diff. + +The cell goes red if *any* tier's probe fails the minimal shape +check; the matrix's per-cell aggregator handles that automatically. +Three tiers run sequentially because count_tokens is cheap (<100ms +per request typical) and the parallelization that matters for the +CLI rows isn't useful here. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_count_tokens_shape, + probe_count_tokens, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +ANTHROPIC_MODELS = [ + "claude-haiku-4-5", + "claude-sonnet-4-6", + "claude-opus-4-7", +] + + +def test_count_tokens_anthropic(compat_result): + """Probe `/v1/messages/count_tokens` for each Anthropic tier and + assert the response shape.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in ANTHROPIC_MODELS: + result = probe_count_tokens( + base_url=base_url, api_key=api_key, model=model + ) + shape_error = assert_count_tokens_shape(result) + if shape_error is not None: + error = f"[{model}] count_tokens probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/count_tokens/test_azure.py b/tests/claude_code/count_tokens/test_azure.py new file mode 100644 index 00000000000..6e67d36ce50 --- /dev/null +++ b/tests/claude_code/count_tokens/test_azure.py @@ -0,0 +1,94 @@ +"""count_tokens x Azure (Microsoft Foundry). + +HTTP-probe row. Unlike the CLI-driven rows, this test never invokes +the `claude` CLI: it `POST`s directly to +`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts +the response is shaped `{"input_tokens": }`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/count_tokens/test_azure.py + ^^^^^^^^^^^^ ^^^^^ + feature_id provider + +Why HTTP probe instead of CLI: + +Claude Code calls `count_tokens` internally to compute budget / +context-window usage display, but the result is consumed by the CLI +in-process and never appears in stream-json events. There is no CLI +flag that emits the count to stdout in a way our existing +stream-json parser can pick up, so we can't test the endpoint round +trip through the CLI surface. + +The proxy *is* expected to expose `/v1/messages/count_tokens` for +every Claude-style provider it routes to -- LiteLLM has historically +had provider-specific bugs in this endpoint (Vertex AI `count_tokens` +returned 400 to proxy gateways; see Claude Code release notes 2.1.121). +Treating it as a matrix row keeps regressions in the cron's daily +diff. + +The cell goes red if *any* tier's probe fails the minimal shape +check; the matrix's per-cell aggregator handles that automatically. +Three tiers run sequentially because count_tokens is cheap (<100ms +per request typical) and the parallelization that matters for the +CLI rows isn't useful here. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_count_tokens_shape, + probe_count_tokens, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +AZURE_MODELS = [ + "claude-haiku-4-5-azure", + "claude-sonnet-4-6-azure", + "claude-opus-4-7-azure", +] + + +def test_count_tokens_azure(compat_result): + """Probe `/v1/messages/count_tokens` for each Azure (Microsoft Foundry) tier and + assert the response shape.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in AZURE_MODELS: + result = probe_count_tokens( + base_url=base_url, api_key=api_key, model=model + ) + shape_error = assert_count_tokens_shape(result) + if shape_error is not None: + error = f"[{model}] count_tokens probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/count_tokens/test_bedrock_converse.py b/tests/claude_code/count_tokens/test_bedrock_converse.py new file mode 100644 index 00000000000..04e7a5c71d0 --- /dev/null +++ b/tests/claude_code/count_tokens/test_bedrock_converse.py @@ -0,0 +1,94 @@ +"""count_tokens x Bedrock (Converse). + +HTTP-probe row. Unlike the CLI-driven rows, this test never invokes +the `claude` CLI: it `POST`s directly to +`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts +the response is shaped `{"input_tokens": }`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/count_tokens/test_bedrock_converse.py + ^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI: + +Claude Code calls `count_tokens` internally to compute budget / +context-window usage display, but the result is consumed by the CLI +in-process and never appears in stream-json events. There is no CLI +flag that emits the count to stdout in a way our existing +stream-json parser can pick up, so we can't test the endpoint round +trip through the CLI surface. + +The proxy *is* expected to expose `/v1/messages/count_tokens` for +every Claude-style provider it routes to -- LiteLLM has historically +had provider-specific bugs in this endpoint (Vertex AI `count_tokens` +returned 400 to proxy gateways; see Claude Code release notes 2.1.121). +Treating it as a matrix row keeps regressions in the cron's daily +diff. + +The cell goes red if *any* tier's probe fails the minimal shape +check; the matrix's per-cell aggregator handles that automatically. +Three tiers run sequentially because count_tokens is cheap (<100ms +per request typical) and the parallelization that matters for the +CLI rows isn't useful here. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_count_tokens_shape, + probe_count_tokens, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_CONVERSE_MODELS = [ + "claude-haiku-4-5-bedrock-converse", + "claude-sonnet-4-6-bedrock-converse", + "claude-opus-4-7-bedrock-converse", +] + + +def test_count_tokens_bedrock_converse(compat_result): + """Probe `/v1/messages/count_tokens` for each Bedrock (Converse) tier and + assert the response shape.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in BEDROCK_CONVERSE_MODELS: + result = probe_count_tokens( + base_url=base_url, api_key=api_key, model=model + ) + shape_error = assert_count_tokens_shape(result) + if shape_error is not None: + error = f"[{model}] count_tokens probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/count_tokens/test_bedrock_invoke.py b/tests/claude_code/count_tokens/test_bedrock_invoke.py new file mode 100644 index 00000000000..f0874919fbc --- /dev/null +++ b/tests/claude_code/count_tokens/test_bedrock_invoke.py @@ -0,0 +1,94 @@ +"""count_tokens x Bedrock (Invoke). + +HTTP-probe row. Unlike the CLI-driven rows, this test never invokes +the `claude` CLI: it `POST`s directly to +`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts +the response is shaped `{"input_tokens": }`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/count_tokens/test_bedrock_invoke.py + ^^^^^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI: + +Claude Code calls `count_tokens` internally to compute budget / +context-window usage display, but the result is consumed by the CLI +in-process and never appears in stream-json events. There is no CLI +flag that emits the count to stdout in a way our existing +stream-json parser can pick up, so we can't test the endpoint round +trip through the CLI surface. + +The proxy *is* expected to expose `/v1/messages/count_tokens` for +every Claude-style provider it routes to -- LiteLLM has historically +had provider-specific bugs in this endpoint (Vertex AI `count_tokens` +returned 400 to proxy gateways; see Claude Code release notes 2.1.121). +Treating it as a matrix row keeps regressions in the cron's daily +diff. + +The cell goes red if *any* tier's probe fails the minimal shape +check; the matrix's per-cell aggregator handles that automatically. +Three tiers run sequentially because count_tokens is cheap (<100ms +per request typical) and the parallelization that matters for the +CLI rows isn't useful here. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_count_tokens_shape, + probe_count_tokens, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_INVOKE_MODELS = [ + "claude-haiku-4-5-bedrock-invoke", + "claude-sonnet-4-6-bedrock-invoke", + "claude-opus-4-7-bedrock-invoke", +] + + +def test_count_tokens_bedrock_invoke(compat_result): + """Probe `/v1/messages/count_tokens` for each Bedrock (Invoke) tier and + assert the response shape.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in BEDROCK_INVOKE_MODELS: + result = probe_count_tokens( + base_url=base_url, api_key=api_key, model=model + ) + shape_error = assert_count_tokens_shape(result) + if shape_error is not None: + error = f"[{model}] count_tokens probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/count_tokens/test_vertex_ai.py b/tests/claude_code/count_tokens/test_vertex_ai.py new file mode 100644 index 00000000000..fe84372cc41 --- /dev/null +++ b/tests/claude_code/count_tokens/test_vertex_ai.py @@ -0,0 +1,94 @@ +"""count_tokens x Vertex AI. + +HTTP-probe row. Unlike the CLI-driven rows, this test never invokes +the `claude` CLI: it `POST`s directly to +`{proxy}/v1/messages/count_tokens` for each Claude tier and asserts +the response is shaped `{"input_tokens": }`. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/count_tokens/test_vertex_ai.py + ^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI: + +Claude Code calls `count_tokens` internally to compute budget / +context-window usage display, but the result is consumed by the CLI +in-process and never appears in stream-json events. There is no CLI +flag that emits the count to stdout in a way our existing +stream-json parser can pick up, so we can't test the endpoint round +trip through the CLI surface. + +The proxy *is* expected to expose `/v1/messages/count_tokens` for +every Claude-style provider it routes to -- LiteLLM has historically +had provider-specific bugs in this endpoint (Vertex AI `count_tokens` +returned 400 to proxy gateways; see Claude Code release notes 2.1.121). +Treating it as a matrix row keeps regressions in the cron's daily +diff. + +The cell goes red if *any* tier's probe fails the minimal shape +check; the matrix's per-cell aggregator handles that automatically. +Three tiers run sequentially because count_tokens is cheap (<100ms +per request typical) and the parallelization that matters for the +CLI rows isn't useful here. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_count_tokens_shape, + probe_count_tokens, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +VERTEX_AI_MODELS = [ + "claude-haiku-4-5-vertex", + "claude-sonnet-4-6-vertex", + "claude-opus-4-7-vertex", +] + + +def test_count_tokens_vertex_ai(compat_result): + """Probe `/v1/messages/count_tokens` for each Vertex AI tier and + assert the response shape.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in VERTEX_AI_MODELS: + result = probe_count_tokens( + base_url=base_url, api_key=api_key, model=model + ) + shape_error = assert_count_tokens_shape(result) + if shape_error is not None: + error = f"[{model}] count_tokens probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/http_probe.py b/tests/claude_code/http_probe.py new file mode 100644 index 00000000000..5628f7af3d9 --- /dev/null +++ b/tests/claude_code/http_probe.py @@ -0,0 +1,265 @@ +"""Direct HTTP probe helpers for the Claude Code compatibility matrix. + +Most matrix cells drive the `claude` CLI in headless mode and observe +the stream-json wire (see `cli_driver.py`). A handful of features the +proxy must support don't have any CLI surface area -- `count_tokens` is +the canonical example: Claude Code calls it internally for budget +display, but the result never appears in stream-json events, so a CLI +test cannot observe whether the endpoint round-tripped correctly +through the proxy for any given provider. + +This module is the second test pattern the matrix supports: a plain +HTTP POST against a LiteLLM proxy endpoint, parsed and shape-checked +in the test, with the same `compat_result` recording convention as the +CLI-driven cells. The goal is to keep this pattern *narrow* -- if a +feature can be tested via the CLI, it should be, because the CLI path +is closer to what real Claude Code users hit. HTTP probes are only for +features the CLI can't reach. + +The probe deliberately uses a short timeout (30s) and small payloads: +this is a "did the request shape survive the proxy's +provider-specific transformations" test, not a load test, and a real +endpoint regression typically surfaces in well under a second of wall +time (400 / 500 from the upstream, or LiteLLM 500 on a transformation +bug). +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass +from typing import Any, Mapping, Optional + +import httpx + + +DEFAULT_TIMEOUT_SECONDS = 30.0 + + +@dataclass +class ProbeResult: + """Structured outcome of a single HTTP probe. + + `status_code` and `body` are the wire response; `payload` is the + parsed JSON body if the response was JSON, else None. Tests assert + on `status_code` + `payload` shape; `body` is preserved so failure + diagnostics can echo the raw error string (which is the only thing + a maintainer needs to triage a red cell). + """ + + status_code: int + body: str + payload: Optional[Mapping[str, Any]] = None + error: Optional[str] = None + + +def probe_count_tokens( + *, + base_url: str, + api_key: str, + model: str, + message: str = "hello world", + timeout: float = DEFAULT_TIMEOUT_SECONDS, +) -> ProbeResult: + """POST to `{base_url}/v1/messages/count_tokens` for `model` and return the parsed result. + + The Anthropic / LiteLLM `count_tokens` endpoint accepts a request + body whose shape mirrors `/v1/messages` (model + messages), and + returns `{"input_tokens": N}` for a successful response. Anything + else -- non-200 status, non-JSON body, missing/non-int + `input_tokens` -- is a regression we want the cell to flip red on. + """ + url = base_url.rstrip("/") + "/v1/messages/count_tokens" + try: + response = httpx.post( + url, + headers={ + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + # `anthropic-version` is required by Anthropic's native + # API and harmless on every other provider the proxy + # routes to. Matches what the Claude Code CLI sends + # for its own internal `count_tokens` calls. + "anthropic-version": "2023-06-01", + }, + json={"model": model, "messages": [{"role": "user", "content": message}]}, + timeout=timeout, + ) + except httpx.HTTPError as exc: + return ProbeResult(status_code=0, body="", error=f"transport: {exc}") + + body = response.text or "" + try: + payload = response.json() if body else None + except (json.JSONDecodeError, ValueError): + payload = None + + return ProbeResult( + status_code=response.status_code, + body=body, + payload=payload, + ) + + +def probe_tool_search( + *, + base_url: str, + api_key: str, + model: str, + timeout: float = DEFAULT_TIMEOUT_SECONDS, +) -> ProbeResult: + """POST to `{base_url}/v1/messages` with a `tool_search_tool_regex_20251119` + tool definition and return the result. + + The shape of the tools array is the one Claude Code emits when its + MCP-tool-search beta is active: a `tool_search_tool_regex_20251119` + discovery tool (name `tool_search_tool_regex`) plus at least one + regular user tool to be searched. LiteLLM's + `is_tool_search_used` helper keys on the `_20251119`-suffixed type + string to decide whether to attach the provider-specific tool-search + beta header (`advanced-tool-use-2025-11-20` for Anthropic/Azure, + `tool-search-tool-2025-10-19` for Vertex/Bedrock). A proxy + regression in that translation will surface here as a 400 from + the upstream complaining about the tool type or beta header. + + The prompt deliberately does not force a tool call -- the goal is + to verify the *request* round-trips without 400 and produces some + response, not to test whether the model decided to invoke + tool_search. That kind of behavior test would couple this row to + Claude Code's model behavior heuristics, which change weekly. + """ + url = base_url.rstrip("/") + "/v1/messages" + payload = { + "model": model, + "max_tokens": 64, + "messages": [ + { + "role": "user", + "content": ( + "If you have a tool to discover other tools, use it to " + "find one. Otherwise reply with the word 'done'." + ), + } + ], + "tools": [ + # The tool_search discovery tool itself. Type is the SDK- + # version-pinned `_20251119` suffix; name is the canonical + # `tool_search_tool_regex` (no suffix) Anthropic accepts. + # LiteLLM keys its beta-header translation on the type. + { + "type": "tool_search_tool_regex_20251119", + "name": "tool_search_tool_regex", + }, + # A trivial user tool for the discovery tool to potentially + # surface. Without at least one non-search tool the request + # is shape-valid but semantically empty; we include one so + # the wire shape mirrors what real Claude Code sends. + { + "name": "add_numbers", + "description": "Add two integers", + "input_schema": { + "type": "object", + "properties": { + "a": {"type": "integer"}, + "b": {"type": "integer"}, + }, + "required": ["a", "b"], + }, + }, + ], + } + try: + response = httpx.post( + url, + headers={ + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + "anthropic-version": "2023-06-01", + }, + json=payload, + timeout=timeout, + ) + except httpx.HTTPError as exc: + return ProbeResult(status_code=0, body="", error=f"transport: {exc}") + + body = response.text or "" + try: + payload_out = response.json() if body else None + except (json.JSONDecodeError, ValueError): + payload_out = None + + return ProbeResult( + status_code=response.status_code, + body=body, + payload=payload_out, + ) + + +def assert_tool_search_shape(result: ProbeResult) -> Optional[str]: + """Return None on success, else describe the first violation. + + Acceptance criteria: + + 1. HTTP status is 200 (no 400 from the upstream rejecting the + tool_search tool type or a missing beta header). + 2. Body is valid JSON. + 3. Body has either `content` (Anthropic-shape passthrough) or + `choices` (LiteLLM normalized openai-shape, used by Bedrock + Converse). Either is acceptable -- the matrix cares that the + proxy *accepts and forwards* tool_search, not that the model + actually chose to invoke it. Tool-invocation behavior is a + model decision the matrix has no business asserting on. + + The cell goes red when the upstream rejects the tool type, the + proxy drops the beta header, or the response shape is unusable. + Anything else (model decided to call or not call tool_search) is + irrelevant for this row. + """ + if result.error is not None: + return f"transport error: {result.error}" + if result.status_code != 200: + return f"status {result.status_code}: {result.body[:400]}" + if result.payload is None: + return f"non-JSON body: {result.body[:400]}" + if not isinstance(result.payload, Mapping): + return f"body is not a JSON object: {type(result.payload).__name__}" + # LiteLLM normalizes some provider responses to OpenAI shape + # (`choices`) and passes others through Anthropic-shape (`content`). + # Accept either; both prove the proxy round-tripped the request. + if "content" not in result.payload and "choices" not in result.payload: + return ( + f"response has neither `content` nor `choices`: " + f"keys={sorted(result.payload.keys())}" + ) + return None + + +def assert_count_tokens_shape(result: ProbeResult) -> Optional[str]: + """Return None on success, or an error string describing the first violation. + + Acceptance criteria are intentionally minimal: + + 1. HTTP status is 200. + 2. Body is valid JSON. + 3. Body has an `input_tokens` key whose value is a positive int. + + Anything beyond that (cache token fields, server metadata) is + optional and varies by provider/transport. Asserting on extras + would create a brittle test that flips red on neutral protocol + drift; matrix cells should only go red on functional regressions + a Claude Code user would feel. + """ + if result.error is not None: + return f"transport error: {result.error}" + if result.status_code != 200: + return f"status {result.status_code}: {result.body[:400]}" + if result.payload is None: + return f"non-JSON body: {result.body[:400]}" + if not isinstance(result.payload, Mapping): + return f"body is not a JSON object: {type(result.payload).__name__}" + tokens = result.payload.get("input_tokens") + if not isinstance(tokens, int) or isinstance(tokens, bool): + return f"input_tokens missing or not an int: got {tokens!r}" + if tokens <= 0: + return f"input_tokens must be positive; got {tokens}" + return None diff --git a/tests/claude_code/long_context_1m/__init__.py b/tests/claude_code/long_context_1m/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/long_context_1m/test_anthropic.py b/tests/claude_code/long_context_1m/test_anthropic.py new file mode 100644 index 00000000000..e331a377fb6 --- /dev/null +++ b/tests/claude_code/long_context_1m/test_anthropic.py @@ -0,0 +1,224 @@ +"""long_context_1m x Anthropic. + +Drive the real `claude` CLI in headless mode with a ~210k-token padded +prompt and the `--betas context-1m-2025-08-07` beta header, route +through a LiteLLM proxy aimed at Anthropic, and assert the request +round-trips: no 400 from a stripped beta header, no 413 from a body +the proxy refused to forward, and a non-empty assistant reply at the +end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/long_context_1m/test_anthropic.py + ^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Cost note (read this before scaling the prompt up): + +This row genuinely exercises the long-context path -- a 210k-token +prompt is *just over* Claude's standard 200k context window, which is +the threshold that requires the `context-1m-2025-08-07` beta header +to be honored end-to-end. Anything shorter would only test whether +the proxy forwards the beta header byte-for-byte; it would not catch +provider-side regressions where the header is forwarded but the +upstream silently truncates beyond the standard context (we've seen +this on third-party gateways). Anything longer is wasted spend. + +Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus. +Daily cost across all five providers (this row only): ~$19. + +Haiku 4.5 is intentionally omitted: it does not support 1M context +(its window is 200k). Reporting `not_applicable` for Haiku would +flip the entire cell to `not_applicable`, hiding genuine 1M +regressions on Sonnet/Opus; instead we exclude Haiku from the model +list entirely and let the matrix's per-cell aggregator green the +cell on Sonnet + Opus passing. This is the one row where the "all +three tiers must pass" rule is relaxed; it's relaxed structurally +(via the model list), not semantically (via not_applicable), so the +matrix builder stays unmodified. + +The prompt is delivered via subprocess stdin rather than a positional +argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt +fits within that comfortably, but stdin is safer (no shell escaping +surprises, no surprise ARG_MAX clamp on a tightened sandbox) and +keeps the driver's `extra_args` slot free for the `--betas` flag. + +`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test +that loops three Claude tiers in a single cell can't accidentally +spend more than ~$18 on this cell. The cap is twice the expected +worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a +provider's pricing changes and the matrix starts spending more than +$10/day on this row. +""" + +from __future__ import annotations + +import os +from typing import Sequence + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the +# 1M-context beta. See module docstring for the per-cell-aggregator +# rationale. +ANTHROPIC_MODELS: Sequence[str] = ( + "claude-sonnet-4-6", + "claude-opus-4-7", +) + +# Beta header that opts an Anthropic-shape model into the 1M context +# window. Same string is accepted on Bedrock (Invoke + Converse) and +# Vertex per LiteLLM's transformers -- no per-provider translation is +# needed for this header, unlike `advanced-tool-use-2025-11-20` / +# `tool-search-tool-2025-10-19`. +LONG_CONTEXT_BETA = "context-1m-2025-08-07" + +# Target a padded prompt that lands just above Claude's standard 200k +# context window so the request can only succeed if the +# `context-1m-2025-08-07` beta header survives all the way to the +# upstream. Below 200k the cell would silently pass even with a +# proxy-dropped beta header; above ~220k we're paying for tokens that +# don't add signal. +TARGET_INPUT_TOKENS = 210_000 + +# Anthropic's English tokenizer averages ~4 chars/token. We cycle +# through several benign pangrams + filler so the padding looks like a +# real document, not a repeating monolith. Identical-line padding + +# "ignore everything above" trips Opus 4.7's safety filter as a +# suspected prompt-injection attempt -- we hit that during smoke +# testing and the cell flipped red for the wrong reason. Varied prose +# with a natural document-style framing keeps the filter quiet. +_PAD_CHUNKS = ( + "The quick brown fox jumps over the lazy dog. ", + "She sells seashells by the seashore on Sunday mornings. ", + "Pack my box with five dozen liquor jugs for the journey. ", + "How vexingly quick daft zebras jump over fences at dawn. ", + "Sphinx of black quartz, judge my vow of silence and patience. ", + "Waltz, bad nymph, for quick jigs in the moonlit meadow. ", + "Glib jocks quiz nymph to vex dwarf with a riddle of stone. ", + "Crazy Fredrick bought many very exquisite opal jewels lately. ", +) +_CHARS_PER_TOKEN = 4 + + +def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: + """Build a ~target_tokens-token padded prompt with a trailing instruction. + + Framing: + + - Lead with a benign document-style preamble that justifies the + long context (so safety filters see the prompt as "long + document review" rather than "adversarial padding"). + - Cycle through a small set of pangrams + filler sentences for + the bulk of the padding. Variety matters: identical repeated + lines look like a denial-of-service or injection attempt to + Anthropic's content filter on the larger tiers. + - End with the actual question. Claude's instruction-following + is stronger on recent tokens, so a 210k-token-into-the-past + instruction would risk a false-fail where the model ignores + it. + + `target_tokens` is an approximation: actual token count depends + on the tokenizer, but Anthropic's English tokenizer averages + ~4 chars/token, so 4 × target_tokens chars of padding gets us + close enough to the 1M-beta threshold (200k) that the proxy's + beta-header handling is the only path to success. + """ + preamble = ( + "I'm going to share an excerpt from a long document with you. " + "It contains a mix of practice sentences a typist might use to " + "warm up; treat the bulk of the text as background context. " + "I'll ask a short question at the end.\n\n" + "Begin excerpt:\n\n" + ) + closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'." + + pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing) + pad_lines = [] + pad_len = 0 + idx = 0 + while pad_len < pad_target_chars: + chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)] + pad_lines.append(chunk) + pad_len += len(chunk) + idx += 1 + return preamble + "".join(pad_lines) + closing + + +def test_long_context_1m_anthropic(compat_result): + """Drive the `claude` CLI with a ~210k-token prompt and the + `context-1m-2025-08-07` beta header; assert no 400 / 413 and a + non-empty reply for Sonnet + Opus.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + long_prompt = _build_long_prompt() + + outcomes = run_claude_models_parallel( + models=ANTHROPIC_MODELS, + prompt=None, + stdin_input=long_prompt, + base_url=base_url, + api_key=api_key, + extra_args=[ + "--betas", + LONG_CONTEXT_BETA, + # Hard ceiling so a runaway test cannot blow the budget. + # See module docstring for sizing. + "--max-budget-usd", + "6", + ], + # Long-context requests can take a couple of minutes on a + # loaded upstream; the driver's default 120s is too tight. + timeout=300.0, + ) + + failures = [] + for model in ANTHROPIC_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if not outcome.text.strip(): + error = f"[{model}] claude returned empty assistant text" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/long_context_1m/test_azure.py b/tests/claude_code/long_context_1m/test_azure.py new file mode 100644 index 00000000000..8119a15d588 --- /dev/null +++ b/tests/claude_code/long_context_1m/test_azure.py @@ -0,0 +1,224 @@ +"""long_context_1m x Azure (Microsoft Foundry). + +Drive the real `claude` CLI in headless mode with a ~210k-token padded +prompt and the `--betas context-1m-2025-08-07` beta header, route +through a LiteLLM proxy aimed at Anthropic, and assert the request +round-trips: no 400 from a stripped beta header, no 413 from a body +the proxy refused to forward, and a non-empty assistant reply at the +end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/long_context_1m/test_azure.py + ^^^^^^^^^^^^^^^ ^^^^^ + feature_id provider + +Cost note (read this before scaling the prompt up): + +This row genuinely exercises the long-context path -- a 210k-token +prompt is *just over* Claude's standard 200k context window, which is +the threshold that requires the `context-1m-2025-08-07` beta header +to be honored end-to-end. Anything shorter would only test whether +the proxy forwards the beta header byte-for-byte; it would not catch +provider-side regressions where the header is forwarded but the +upstream silently truncates beyond the standard context (we've seen +this on third-party gateways). Anything longer is wasted spend. + +Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus. +Daily cost across all five providers (this row only): ~$19. + +Haiku 4.5 is intentionally omitted: it does not support 1M context +(its window is 200k). Reporting `not_applicable` for Haiku would +flip the entire cell to `not_applicable`, hiding genuine 1M +regressions on Sonnet/Opus; instead we exclude Haiku from the model +list entirely and let the matrix's per-cell aggregator green the +cell on Sonnet + Opus passing. This is the one row where the "all +three tiers must pass" rule is relaxed; it's relaxed structurally +(via the model list), not semantically (via not_applicable), so the +matrix builder stays unmodified. + +The prompt is delivered via subprocess stdin rather than a positional +argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt +fits within that comfortably, but stdin is safer (no shell escaping +surprises, no surprise ARG_MAX clamp on a tightened sandbox) and +keeps the driver's `extra_args` slot free for the `--betas` flag. + +`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test +that loops three Claude tiers in a single cell can't accidentally +spend more than ~$18 on this cell. The cap is twice the expected +worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a +provider's pricing changes and the matrix starts spending more than +$10/day on this row. +""" + +from __future__ import annotations + +import os +from typing import Sequence + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the +# 1M-context beta. See module docstring for the per-cell-aggregator +# rationale. +AZURE_MODELS: Sequence[str] = ( + "claude-sonnet-4-6-azure", + "claude-opus-4-7-azure", +) + +# Beta header that opts an Anthropic-shape model into the 1M context +# window. Same string is accepted on Bedrock (Invoke + Converse) and +# Vertex per LiteLLM's transformers -- no per-provider translation is +# needed for this header, unlike `advanced-tool-use-2025-11-20` / +# `tool-search-tool-2025-10-19`. +LONG_CONTEXT_BETA = "context-1m-2025-08-07" + +# Target a padded prompt that lands just above Claude's standard 200k +# context window so the request can only succeed if the +# `context-1m-2025-08-07` beta header survives all the way to the +# upstream. Below 200k the cell would silently pass even with a +# proxy-dropped beta header; above ~220k we're paying for tokens that +# don't add signal. +TARGET_INPUT_TOKENS = 210_000 + +# Anthropic's English tokenizer averages ~4 chars/token. We cycle +# through several benign pangrams + filler so the padding looks like a +# real document, not a repeating monolith. Identical-line padding + +# "ignore everything above" trips Opus 4.7's safety filter as a +# suspected prompt-injection attempt -- we hit that during smoke +# testing and the cell flipped red for the wrong reason. Varied prose +# with a natural document-style framing keeps the filter quiet. +_PAD_CHUNKS = ( + "The quick brown fox jumps over the lazy dog. ", + "She sells seashells by the seashore on Sunday mornings. ", + "Pack my box with five dozen liquor jugs for the journey. ", + "How vexingly quick daft zebras jump over fences at dawn. ", + "Sphinx of black quartz, judge my vow of silence and patience. ", + "Waltz, bad nymph, for quick jigs in the moonlit meadow. ", + "Glib jocks quiz nymph to vex dwarf with a riddle of stone. ", + "Crazy Fredrick bought many very exquisite opal jewels lately. ", +) +_CHARS_PER_TOKEN = 4 + + +def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: + """Build a ~target_tokens-token padded prompt with a trailing instruction. + + Framing: + + - Lead with a benign document-style preamble that justifies the + long context (so safety filters see the prompt as "long + document review" rather than "adversarial padding"). + - Cycle through a small set of pangrams + filler sentences for + the bulk of the padding. Variety matters: identical repeated + lines look like a denial-of-service or injection attempt to + Anthropic's content filter on the larger tiers. + - End with the actual question. Claude's instruction-following + is stronger on recent tokens, so a 210k-token-into-the-past + instruction would risk a false-fail where the model ignores + it. + + `target_tokens` is an approximation: actual token count depends + on the tokenizer, but Anthropic's English tokenizer averages + ~4 chars/token, so 4 × target_tokens chars of padding gets us + close enough to the 1M-beta threshold (200k) that the proxy's + beta-header handling is the only path to success. + """ + preamble = ( + "I'm going to share an excerpt from a long document with you. " + "It contains a mix of practice sentences a typist might use to " + "warm up; treat the bulk of the text as background context. " + "I'll ask a short question at the end.\n\n" + "Begin excerpt:\n\n" + ) + closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'." + + pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing) + pad_lines = [] + pad_len = 0 + idx = 0 + while pad_len < pad_target_chars: + chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)] + pad_lines.append(chunk) + pad_len += len(chunk) + idx += 1 + return preamble + "".join(pad_lines) + closing + + +def test_long_context_1m_azure(compat_result): + """Drive the `claude` CLI (Azure (Microsoft Foundry)) with a ~210k-token prompt and the + `context-1m-2025-08-07` beta header; assert no 400 / 413 and a + non-empty reply for Sonnet + Opus.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + long_prompt = _build_long_prompt() + + outcomes = run_claude_models_parallel( + models=AZURE_MODELS, + prompt=None, + stdin_input=long_prompt, + base_url=base_url, + api_key=api_key, + extra_args=[ + "--betas", + LONG_CONTEXT_BETA, + # Hard ceiling so a runaway test cannot blow the budget. + # See module docstring for sizing. + "--max-budget-usd", + "6", + ], + # Long-context requests can take a couple of minutes on a + # loaded upstream; the driver's default 120s is too tight. + timeout=300.0, + ) + + failures = [] + for model in AZURE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if not outcome.text.strip(): + error = f"[{model}] claude returned empty assistant text" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/long_context_1m/test_bedrock_converse.py b/tests/claude_code/long_context_1m/test_bedrock_converse.py new file mode 100644 index 00000000000..4fc335b2aa1 --- /dev/null +++ b/tests/claude_code/long_context_1m/test_bedrock_converse.py @@ -0,0 +1,224 @@ +"""long_context_1m x Bedrock (Converse). + +Drive the real `claude` CLI in headless mode with a ~210k-token padded +prompt and the `--betas context-1m-2025-08-07` beta header, route +through a LiteLLM proxy aimed at Anthropic, and assert the request +round-trips: no 400 from a stripped beta header, no 413 from a body +the proxy refused to forward, and a non-empty assistant reply at the +end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/long_context_1m/test_bedrock_converse.py + ^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider + +Cost note (read this before scaling the prompt up): + +This row genuinely exercises the long-context path -- a 210k-token +prompt is *just over* Claude's standard 200k context window, which is +the threshold that requires the `context-1m-2025-08-07` beta header +to be honored end-to-end. Anything shorter would only test whether +the proxy forwards the beta header byte-for-byte; it would not catch +provider-side regressions where the header is forwarded but the +upstream silently truncates beyond the standard context (we've seen +this on third-party gateways). Anything longer is wasted spend. + +Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus. +Daily cost across all five providers (this row only): ~$19. + +Haiku 4.5 is intentionally omitted: it does not support 1M context +(its window is 200k). Reporting `not_applicable` for Haiku would +flip the entire cell to `not_applicable`, hiding genuine 1M +regressions on Sonnet/Opus; instead we exclude Haiku from the model +list entirely and let the matrix's per-cell aggregator green the +cell on Sonnet + Opus passing. This is the one row where the "all +three tiers must pass" rule is relaxed; it's relaxed structurally +(via the model list), not semantically (via not_applicable), so the +matrix builder stays unmodified. + +The prompt is delivered via subprocess stdin rather than a positional +argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt +fits within that comfortably, but stdin is safer (no shell escaping +surprises, no surprise ARG_MAX clamp on a tightened sandbox) and +keeps the driver's `extra_args` slot free for the `--betas` flag. + +`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test +that loops three Claude tiers in a single cell can't accidentally +spend more than ~$18 on this cell. The cap is twice the expected +worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a +provider's pricing changes and the matrix starts spending more than +$10/day on this row. +""" + +from __future__ import annotations + +import os +from typing import Sequence + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the +# 1M-context beta. See module docstring for the per-cell-aggregator +# rationale. +BEDROCK_CONVERSE_MODELS: Sequence[str] = ( + "claude-sonnet-4-6-bedrock-converse", + "claude-opus-4-7-bedrock-converse", +) + +# Beta header that opts an Anthropic-shape model into the 1M context +# window. Same string is accepted on Bedrock (Invoke + Converse) and +# Vertex per LiteLLM's transformers -- no per-provider translation is +# needed for this header, unlike `advanced-tool-use-2025-11-20` / +# `tool-search-tool-2025-10-19`. +LONG_CONTEXT_BETA = "context-1m-2025-08-07" + +# Target a padded prompt that lands just above Claude's standard 200k +# context window so the request can only succeed if the +# `context-1m-2025-08-07` beta header survives all the way to the +# upstream. Below 200k the cell would silently pass even with a +# proxy-dropped beta header; above ~220k we're paying for tokens that +# don't add signal. +TARGET_INPUT_TOKENS = 210_000 + +# Anthropic's English tokenizer averages ~4 chars/token. We cycle +# through several benign pangrams + filler so the padding looks like a +# real document, not a repeating monolith. Identical-line padding + +# "ignore everything above" trips Opus 4.7's safety filter as a +# suspected prompt-injection attempt -- we hit that during smoke +# testing and the cell flipped red for the wrong reason. Varied prose +# with a natural document-style framing keeps the filter quiet. +_PAD_CHUNKS = ( + "The quick brown fox jumps over the lazy dog. ", + "She sells seashells by the seashore on Sunday mornings. ", + "Pack my box with five dozen liquor jugs for the journey. ", + "How vexingly quick daft zebras jump over fences at dawn. ", + "Sphinx of black quartz, judge my vow of silence and patience. ", + "Waltz, bad nymph, for quick jigs in the moonlit meadow. ", + "Glib jocks quiz nymph to vex dwarf with a riddle of stone. ", + "Crazy Fredrick bought many very exquisite opal jewels lately. ", +) +_CHARS_PER_TOKEN = 4 + + +def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: + """Build a ~target_tokens-token padded prompt with a trailing instruction. + + Framing: + + - Lead with a benign document-style preamble that justifies the + long context (so safety filters see the prompt as "long + document review" rather than "adversarial padding"). + - Cycle through a small set of pangrams + filler sentences for + the bulk of the padding. Variety matters: identical repeated + lines look like a denial-of-service or injection attempt to + Anthropic's content filter on the larger tiers. + - End with the actual question. Claude's instruction-following + is stronger on recent tokens, so a 210k-token-into-the-past + instruction would risk a false-fail where the model ignores + it. + + `target_tokens` is an approximation: actual token count depends + on the tokenizer, but Anthropic's English tokenizer averages + ~4 chars/token, so 4 × target_tokens chars of padding gets us + close enough to the 1M-beta threshold (200k) that the proxy's + beta-header handling is the only path to success. + """ + preamble = ( + "I'm going to share an excerpt from a long document with you. " + "It contains a mix of practice sentences a typist might use to " + "warm up; treat the bulk of the text as background context. " + "I'll ask a short question at the end.\n\n" + "Begin excerpt:\n\n" + ) + closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'." + + pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing) + pad_lines = [] + pad_len = 0 + idx = 0 + while pad_len < pad_target_chars: + chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)] + pad_lines.append(chunk) + pad_len += len(chunk) + idx += 1 + return preamble + "".join(pad_lines) + closing + + +def test_long_context_1m_bedrock_converse(compat_result): + """Drive the `claude` CLI (Bedrock (Converse)) with a ~210k-token prompt and the + `context-1m-2025-08-07` beta header; assert no 400 / 413 and a + non-empty reply for Sonnet + Opus.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + long_prompt = _build_long_prompt() + + outcomes = run_claude_models_parallel( + models=BEDROCK_CONVERSE_MODELS, + prompt=None, + stdin_input=long_prompt, + base_url=base_url, + api_key=api_key, + extra_args=[ + "--betas", + LONG_CONTEXT_BETA, + # Hard ceiling so a runaway test cannot blow the budget. + # See module docstring for sizing. + "--max-budget-usd", + "6", + ], + # Long-context requests can take a couple of minutes on a + # loaded upstream; the driver's default 120s is too tight. + timeout=300.0, + ) + + failures = [] + for model in BEDROCK_CONVERSE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if not outcome.text.strip(): + error = f"[{model}] claude returned empty assistant text" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/long_context_1m/test_bedrock_invoke.py b/tests/claude_code/long_context_1m/test_bedrock_invoke.py new file mode 100644 index 00000000000..0b684fb2a46 --- /dev/null +++ b/tests/claude_code/long_context_1m/test_bedrock_invoke.py @@ -0,0 +1,224 @@ +"""long_context_1m x Bedrock (Invoke). + +Drive the real `claude` CLI in headless mode with a ~210k-token padded +prompt and the `--betas context-1m-2025-08-07` beta header, route +through a LiteLLM proxy aimed at Anthropic, and assert the request +round-trips: no 400 from a stripped beta header, no 413 from a body +the proxy refused to forward, and a non-empty assistant reply at the +end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/long_context_1m/test_bedrock_invoke.py + ^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider + +Cost note (read this before scaling the prompt up): + +This row genuinely exercises the long-context path -- a 210k-token +prompt is *just over* Claude's standard 200k context window, which is +the threshold that requires the `context-1m-2025-08-07` beta header +to be honored end-to-end. Anything shorter would only test whether +the proxy forwards the beta header byte-for-byte; it would not catch +provider-side regressions where the header is forwarded but the +upstream silently truncates beyond the standard context (we've seen +this on third-party gateways). Anything longer is wasted spend. + +Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus. +Daily cost across all five providers (this row only): ~$19. + +Haiku 4.5 is intentionally omitted: it does not support 1M context +(its window is 200k). Reporting `not_applicable` for Haiku would +flip the entire cell to `not_applicable`, hiding genuine 1M +regressions on Sonnet/Opus; instead we exclude Haiku from the model +list entirely and let the matrix's per-cell aggregator green the +cell on Sonnet + Opus passing. This is the one row where the "all +three tiers must pass" rule is relaxed; it's relaxed structurally +(via the model list), not semantically (via not_applicable), so the +matrix builder stays unmodified. + +The prompt is delivered via subprocess stdin rather than a positional +argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt +fits within that comfortably, but stdin is safer (no shell escaping +surprises, no surprise ARG_MAX clamp on a tightened sandbox) and +keeps the driver's `extra_args` slot free for the `--betas` flag. + +`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test +that loops three Claude tiers in a single cell can't accidentally +spend more than ~$18 on this cell. The cap is twice the expected +worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a +provider's pricing changes and the matrix starts spending more than +$10/day on this row. +""" + +from __future__ import annotations + +import os +from typing import Sequence + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the +# 1M-context beta. See module docstring for the per-cell-aggregator +# rationale. +BEDROCK_INVOKE_MODELS: Sequence[str] = ( + "claude-sonnet-4-6-bedrock-invoke", + "claude-opus-4-7-bedrock-invoke", +) + +# Beta header that opts an Anthropic-shape model into the 1M context +# window. Same string is accepted on Bedrock (Invoke + Converse) and +# Vertex per LiteLLM's transformers -- no per-provider translation is +# needed for this header, unlike `advanced-tool-use-2025-11-20` / +# `tool-search-tool-2025-10-19`. +LONG_CONTEXT_BETA = "context-1m-2025-08-07" + +# Target a padded prompt that lands just above Claude's standard 200k +# context window so the request can only succeed if the +# `context-1m-2025-08-07` beta header survives all the way to the +# upstream. Below 200k the cell would silently pass even with a +# proxy-dropped beta header; above ~220k we're paying for tokens that +# don't add signal. +TARGET_INPUT_TOKENS = 210_000 + +# Anthropic's English tokenizer averages ~4 chars/token. We cycle +# through several benign pangrams + filler so the padding looks like a +# real document, not a repeating monolith. Identical-line padding + +# "ignore everything above" trips Opus 4.7's safety filter as a +# suspected prompt-injection attempt -- we hit that during smoke +# testing and the cell flipped red for the wrong reason. Varied prose +# with a natural document-style framing keeps the filter quiet. +_PAD_CHUNKS = ( + "The quick brown fox jumps over the lazy dog. ", + "She sells seashells by the seashore on Sunday mornings. ", + "Pack my box with five dozen liquor jugs for the journey. ", + "How vexingly quick daft zebras jump over fences at dawn. ", + "Sphinx of black quartz, judge my vow of silence and patience. ", + "Waltz, bad nymph, for quick jigs in the moonlit meadow. ", + "Glib jocks quiz nymph to vex dwarf with a riddle of stone. ", + "Crazy Fredrick bought many very exquisite opal jewels lately. ", +) +_CHARS_PER_TOKEN = 4 + + +def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: + """Build a ~target_tokens-token padded prompt with a trailing instruction. + + Framing: + + - Lead with a benign document-style preamble that justifies the + long context (so safety filters see the prompt as "long + document review" rather than "adversarial padding"). + - Cycle through a small set of pangrams + filler sentences for + the bulk of the padding. Variety matters: identical repeated + lines look like a denial-of-service or injection attempt to + Anthropic's content filter on the larger tiers. + - End with the actual question. Claude's instruction-following + is stronger on recent tokens, so a 210k-token-into-the-past + instruction would risk a false-fail where the model ignores + it. + + `target_tokens` is an approximation: actual token count depends + on the tokenizer, but Anthropic's English tokenizer averages + ~4 chars/token, so 4 × target_tokens chars of padding gets us + close enough to the 1M-beta threshold (200k) that the proxy's + beta-header handling is the only path to success. + """ + preamble = ( + "I'm going to share an excerpt from a long document with you. " + "It contains a mix of practice sentences a typist might use to " + "warm up; treat the bulk of the text as background context. " + "I'll ask a short question at the end.\n\n" + "Begin excerpt:\n\n" + ) + closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'." + + pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing) + pad_lines = [] + pad_len = 0 + idx = 0 + while pad_len < pad_target_chars: + chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)] + pad_lines.append(chunk) + pad_len += len(chunk) + idx += 1 + return preamble + "".join(pad_lines) + closing + + +def test_long_context_1m_bedrock_invoke(compat_result): + """Drive the `claude` CLI (Bedrock (Invoke)) with a ~210k-token prompt and the + `context-1m-2025-08-07` beta header; assert no 400 / 413 and a + non-empty reply for Sonnet + Opus.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + long_prompt = _build_long_prompt() + + outcomes = run_claude_models_parallel( + models=BEDROCK_INVOKE_MODELS, + prompt=None, + stdin_input=long_prompt, + base_url=base_url, + api_key=api_key, + extra_args=[ + "--betas", + LONG_CONTEXT_BETA, + # Hard ceiling so a runaway test cannot blow the budget. + # See module docstring for sizing. + "--max-budget-usd", + "6", + ], + # Long-context requests can take a couple of minutes on a + # loaded upstream; the driver's default 120s is too tight. + timeout=300.0, + ) + + failures = [] + for model in BEDROCK_INVOKE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if not outcome.text.strip(): + error = f"[{model}] claude returned empty assistant text" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/long_context_1m/test_vertex_ai.py b/tests/claude_code/long_context_1m/test_vertex_ai.py new file mode 100644 index 00000000000..1afb7d437ab --- /dev/null +++ b/tests/claude_code/long_context_1m/test_vertex_ai.py @@ -0,0 +1,224 @@ +"""long_context_1m x Vertex AI. + +Drive the real `claude` CLI in headless mode with a ~210k-token padded +prompt and the `--betas context-1m-2025-08-07` beta header, route +through a LiteLLM proxy aimed at Anthropic, and assert the request +round-trips: no 400 from a stripped beta header, no 413 from a body +the proxy refused to forward, and a non-empty assistant reply at the +end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/long_context_1m/test_vertex_ai.py + ^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Cost note (read this before scaling the prompt up): + +This row genuinely exercises the long-context path -- a 210k-token +prompt is *just over* Claude's standard 200k context window, which is +the threshold that requires the `context-1m-2025-08-07` beta header +to be honored end-to-end. Anything shorter would only test whether +the proxy forwards the beta header byte-for-byte; it would not catch +provider-side regressions where the header is forwarded but the +upstream silently truncates beyond the standard context (we've seen +this on third-party gateways). Anything longer is wasted spend. + +Per-cell cost at 210k input tokens: ~$0.63 Sonnet, ~$3.15 Opus. +Daily cost across all five providers (this row only): ~$19. + +Haiku 4.5 is intentionally omitted: it does not support 1M context +(its window is 200k). Reporting `not_applicable` for Haiku would +flip the entire cell to `not_applicable`, hiding genuine 1M +regressions on Sonnet/Opus; instead we exclude Haiku from the model +list entirely and let the matrix's per-cell aggregator green the +cell on Sonnet + Opus passing. This is the one row where the "all +three tiers must pass" rule is relaxed; it's relaxed structurally +(via the model list), not semantically (via not_applicable), so the +matrix builder stays unmodified. + +The prompt is delivered via subprocess stdin rather than a positional +argument. ARG_MAX on Linux is typically 2MB and an 840KB prompt +fits within that comfortably, but stdin is safer (no shell escaping +surprises, no surprise ARG_MAX clamp on a tightened sandbox) and +keeps the driver's `extra_args` slot free for the `--betas` flag. + +`--max-budget-usd 6` is a runaway-loop guard: a misbehaving test +that loops three Claude tiers in a single cell can't accidentally +spend more than ~$18 on this cell. The cap is twice the expected +worst case (Opus @ 210k = $3.15) plus a 50% margin. Tighten it if a +provider's pricing changes and the matrix starts spending more than +$10/day on this row. +""" + +from __future__ import annotations + +import os +from typing import Sequence + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the +# 1M-context beta. See module docstring for the per-cell-aggregator +# rationale. +VERTEX_AI_MODELS: Sequence[str] = ( + "claude-sonnet-4-6-vertex", + "claude-opus-4-7-vertex", +) + +# Beta header that opts an Anthropic-shape model into the 1M context +# window. Same string is accepted on Bedrock (Invoke + Converse) and +# Vertex per LiteLLM's transformers -- no per-provider translation is +# needed for this header, unlike `advanced-tool-use-2025-11-20` / +# `tool-search-tool-2025-10-19`. +LONG_CONTEXT_BETA = "context-1m-2025-08-07" + +# Target a padded prompt that lands just above Claude's standard 200k +# context window so the request can only succeed if the +# `context-1m-2025-08-07` beta header survives all the way to the +# upstream. Below 200k the cell would silently pass even with a +# proxy-dropped beta header; above ~220k we're paying for tokens that +# don't add signal. +TARGET_INPUT_TOKENS = 210_000 + +# Anthropic's English tokenizer averages ~4 chars/token. We cycle +# through several benign pangrams + filler so the padding looks like a +# real document, not a repeating monolith. Identical-line padding + +# "ignore everything above" trips Opus 4.7's safety filter as a +# suspected prompt-injection attempt -- we hit that during smoke +# testing and the cell flipped red for the wrong reason. Varied prose +# with a natural document-style framing keeps the filter quiet. +_PAD_CHUNKS = ( + "The quick brown fox jumps over the lazy dog. ", + "She sells seashells by the seashore on Sunday mornings. ", + "Pack my box with five dozen liquor jugs for the journey. ", + "How vexingly quick daft zebras jump over fences at dawn. ", + "Sphinx of black quartz, judge my vow of silence and patience. ", + "Waltz, bad nymph, for quick jigs in the moonlit meadow. ", + "Glib jocks quiz nymph to vex dwarf with a riddle of stone. ", + "Crazy Fredrick bought many very exquisite opal jewels lately. ", +) +_CHARS_PER_TOKEN = 4 + + +def _build_long_prompt(target_tokens: int = TARGET_INPUT_TOKENS) -> str: + """Build a ~target_tokens-token padded prompt with a trailing instruction. + + Framing: + + - Lead with a benign document-style preamble that justifies the + long context (so safety filters see the prompt as "long + document review" rather than "adversarial padding"). + - Cycle through a small set of pangrams + filler sentences for + the bulk of the padding. Variety matters: identical repeated + lines look like a denial-of-service or injection attempt to + Anthropic's content filter on the larger tiers. + - End with the actual question. Claude's instruction-following + is stronger on recent tokens, so a 210k-token-into-the-past + instruction would risk a false-fail where the model ignores + it. + + `target_tokens` is an approximation: actual token count depends + on the tokenizer, but Anthropic's English tokenizer averages + ~4 chars/token, so 4 × target_tokens chars of padding gets us + close enough to the 1M-beta threshold (200k) that the proxy's + beta-header handling is the only path to success. + """ + preamble = ( + "I'm going to share an excerpt from a long document with you. " + "It contains a mix of practice sentences a typist might use to " + "warm up; treat the bulk of the text as background context. " + "I'll ask a short question at the end.\n\n" + "Begin excerpt:\n\n" + ) + closing = "\n\nEnd of excerpt. Please reply with the single word 'ok'." + + pad_target_chars = target_tokens * _CHARS_PER_TOKEN - len(preamble) - len(closing) + pad_lines = [] + pad_len = 0 + idx = 0 + while pad_len < pad_target_chars: + chunk = _PAD_CHUNKS[idx % len(_PAD_CHUNKS)] + pad_lines.append(chunk) + pad_len += len(chunk) + idx += 1 + return preamble + "".join(pad_lines) + closing + + +def test_long_context_1m_vertex_ai(compat_result): + """Drive the `claude` CLI (Vertex AI) with a ~210k-token prompt and the + `context-1m-2025-08-07` beta header; assert no 400 / 413 and a + non-empty reply for Sonnet + Opus.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + long_prompt = _build_long_prompt() + + outcomes = run_claude_models_parallel( + models=VERTEX_AI_MODELS, + prompt=None, + stdin_input=long_prompt, + base_url=base_url, + api_key=api_key, + extra_args=[ + "--betas", + LONG_CONTEXT_BETA, + # Hard ceiling so a runaway test cannot blow the budget. + # See module docstring for sizing. + "--max-budget-usd", + "6", + ], + # Long-context requests can take a couple of minutes on a + # loaded upstream; the driver's default 120s is too tight. + timeout=300.0, + ) + + failures = [] + for model in VERTEX_AI_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if not outcome.text.strip(): + error = f"[{model}] claude returned empty assistant text" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/manifest.yaml b/tests/claude_code/manifest.yaml index ccafe10a904..43cba51c073 100644 --- a/tests/claude_code/manifest.yaml +++ b/tests/claude_code/manifest.yaml @@ -32,8 +32,17 @@ features: name: Prompt caching (5m TTL) - id: vision name: Vision - - id: extended_thinking - name: Extended thinking + - id: thinking + name: Thinking + # The single row covers both API shapes Anthropic exposes — manual + # `thinking: {type: "enabled", budget_tokens: N}` (Haiku 4.5) and + # `thinking: {type: "adaptive"}` (Opus 4.7); Sonnet 4.6 supports + # either and Claude Code picks per model. A break in either + # transformer surfaces as a red cell because all three tiers must + # pass for the cell to go green. The row was named + # `extended_thinking` historically; Anthropic's docs now reserve + # that name for the deprecated manual mode only, so the row was + # renamed to the feature-level "Thinking". - id: tool_use_streaming name: Tool use (streaming / fine-grained) - id: thinking_with_tool_use @@ -44,3 +53,51 @@ features: name: Prompt caching (1h TTL) - id: web_search name: Web search (server tool) + - id: structured_outputs + name: Structured outputs + # Drives `claude --json-schema ''`. Implementation note: + # Claude Code translates `--json-schema` to a synthetic + # `StructuredOutput` tool whose `input_schema` is the user's + # schema, then surfaces the tool_use input as + # `structured_output: {...}` on the trailing `result` event. + # This row tests that proxy-side handling of that tool round- + # trips end-to-end. It does NOT test Anthropic's server-side + # `output_config.schema` parameter (a separate feature used + # internally by Claude Code for session-title generation) -- + # `output_config` regressions surface in the HTTP-probe rows. + - id: count_tokens + name: count_tokens endpoint + # HTTP-probe row. Sends a direct POST to + # `{proxy}/v1/messages/count_tokens` for each Claude tier and + # asserts the response is shaped `{"input_tokens": }`. The CLI uses this endpoint internally but never + # surfaces its result in stream-json, so the only way to test + # the proxy's handling of it is to hit it directly. LiteLLM has + # shipped fixes here (e.g. Claude Code release-notes 2.1.121 + # "Vertex AI count_tokens returning 400 errors for proxy + # gateways"), which is exactly the regression class this row + # is meant to catch. + - id: tool_search + name: Tool search (MCP discovery) + # HTTP-probe row. Sends a request whose `tools` array includes + # a `tool_search_tool_regex_20251119` discovery tool and asserts + # the proxy + upstream accept it. This verifies LiteLLM's + # per-provider beta-header translation + # (`advanced-tool-use-2025-11-20` for Anthropic/Azure, + # `tool-search-tool-2025-10-19` for Vertex/Bedrock) is wired up. + # We deliberately don't try to trigger Claude Code's MCP-fan-out + # heuristic via `--mcp-config` -- that would couple the row to + # an internal behavior threshold that changes between Claude + # Code releases. The HTTP probe hits the bug surface LiteLLM + # has actually shipped fixes for (2.1.117, 2.1.72, 2.1.70 per + # the Claude Code release notes). + - id: long_context_1m + name: Long context (1M) + # Sends a ~210k-token padded prompt with the + # `context-1m-2025-08-07` beta header. Just-above the standard + # 200k context window so the request can only succeed when the + # beta header makes it all the way through the proxy to the + # upstream. Haiku 4.5 is intentionally omitted from this row's + # model list (its window is 200k); Sonnet 4.6 and Opus 4.7 are + # the only tiers exercised. Costs roughly $4/cell/run -- + # tighten the prompt-token target if pricing changes meaningfully. diff --git a/tests/claude_code/run_compat.sh b/tests/claude_code/run_compat.sh index 2e62dc65fb5..d2c79ea2528 100755 --- a/tests/claude_code/run_compat.sh +++ b/tests/claude_code/run_compat.sh @@ -78,7 +78,7 @@ PATH="$HOME/.local/bin:$PATH" \ uv run pytest \ tests/claude_code/basic_messaging_non_streaming \ tests/claude_code/basic_messaging_streaming \ - tests/claude_code/extended_thinking \ + tests/claude_code/thinking \ tests/claude_code/tool_use \ tests/claude_code/vision \ tests/claude_code/prompt_caching_5m \ diff --git a/tests/claude_code/sample_compatibility-matrix.json b/tests/claude_code/sample_compatibility-matrix.json index 04c790e0b44..cfc7f3885d3 100644 --- a/tests/claude_code/sample_compatibility-matrix.json +++ b/tests/claude_code/sample_compatibility-matrix.json @@ -117,8 +117,8 @@ } }, { - "id": "extended_thinking", - "name": "Extended thinking", + "id": "thinking", + "name": "Thinking", "providers": { "anthropic": { "status": "pass" diff --git a/tests/claude_code/structured_outputs/__init__.py b/tests/claude_code/structured_outputs/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/structured_outputs/test_anthropic.py b/tests/claude_code/structured_outputs/test_anthropic.py new file mode 100644 index 00000000000..95683ac93f2 --- /dev/null +++ b/tests/claude_code/structured_outputs/test_anthropic.py @@ -0,0 +1,220 @@ +"""structured_outputs x Anthropic. + +Drive the real `claude` CLI in headless mode with the `--json-schema` +flag, route through a LiteLLM proxy aimed at Anthropic, and assert that +the final stream-json `result` event surfaces a `structured_output` +object whose shape matches the schema. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/structured_outputs/test_anthropic.py + ^^^^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +What this row actually exercises (and what it does not): + +`--json-schema` is implemented client-side by Claude Code: the CLI +synthesizes a synthetic `StructuredOutput` tool whose `input_schema` +equals the user-supplied JSON Schema, forces the model toward it, and +finally extracts the tool_use input on the trailing `result` event as +`structured_output: {...}`. The proxy never sees `output_config.schema` +in this flow -- it sees a normal `tools` array with one synthetic +tool. + +This makes the row a tool-use feature test in disguise. It's still a +distinct row from `tool_use` because: + + - The synthetic tool is generated per request from a user schema, not + a developer-declared one. Provider-side bugs that special-case + `Claude Code`-generated tool names (e.g. case-folding `tool_use` + blocks back to lowercase, or stripping the StructuredOutput-only + `additionalProperties: false`) only surface here. + - The success signal lives on the *final* `result` event, not the + intermediate `assistant` events the `tool_use` row checks. A proxy + that drops trailing events (seen in early Bedrock Converse SSE + plumbing) breaks this cell while leaving `tool_use` green. + +It is NOT a test of Anthropic's server-side `output_config.schema` +parameter -- that's a different feature used internally by Claude Code +for session-title generation and is not reachable from any CLI flag. +LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in +the `count_tokens` and other HTTP-probe rows, not here. + +Three Claude tiers run in parallel; one `compat_result.add(...)` per +tier so the matrix's "all three must pass" rule applies. +""" + +from __future__ import annotations + +import json +import os +import re +from typing import Any, Mapping, Optional, Sequence, Tuple + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +ANTHROPIC_MODELS = [ + "claude-haiku-4-5", + "claude-sonnet-4-6", + "claude-opus-4-7", +] + +# Minimal schema with one required integer field. Kept intentionally +# small -- the matrix tests the *plumbing*, not the model's ability to +# satisfy a complex schema. A trivial arithmetic prompt + a one-field +# integer schema gives every tier (including Haiku) enough headroom +# that schema satisfaction is essentially deterministic, isolating +# failures to the proxy / transport. +SCHEMA = { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + "additionalProperties": False, +} +SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":")) + +# A prompt the model has no reason to misanswer; we don't check the +# value, but a wrong answer would suggest the structured-output +# pathway is silently degrading reasoning, which is itself worth +# noticing. +PROMPT = "What is 2 + 2? Reply only via the structured output." + + +def _extract_structured_output( + events: Sequence[Mapping[str, Any]], +) -> Optional[Mapping[str, Any]]: + """Return the `structured_output` payload from the last `result` event. + + Claude Code emits its terminal stream-json line as + `{"type":"result","structured_output":{...},...}` when a request + used `--json-schema` and the model actually produced a valid tool + call. If the model bailed or the proxy ate the trailing events, + `structured_output` is missing -- which is exactly the failure + mode we want this row to surface, so the caller treats `None` as + "feature did not work end-to-end". + """ + for event in reversed(list(events)): + if event.get("type") != "result": + continue + so = event.get("structured_output") + if isinstance(so, Mapping): + return so + return None + + +def _validate_against_schema( + payload: Mapping[str, Any], schema: Mapping[str, Any] +) -> Optional[str]: + """Tiny shape validator covering the subset we actually need. + + We deliberately do not pull in `jsonschema` as a test dep: the + matrix's success signal is "does the proxy let the synthetic + StructuredOutput tool round-trip end-to-end", and that's + answerable with a presence + type check over `required` keys. + Any malformed schema beyond that would be a Claude Code bug, + not a LiteLLM-proxy bug, so a deeper check would only add false + failures on the wrong axis. + """ + type_map = { + "integer": int, + "number": (int, float), + "string": str, + "boolean": bool, + "array": list, + "object": Mapping, + } + required = schema.get("required") or [] + properties = schema.get("properties") or {} + for key in required: + if key not in payload: + return f"missing required key {key!r}" + expected = (properties.get(key) or {}).get("type") + if expected and expected in type_map: + if not isinstance(payload[key], type_map[expected]): + return ( + f"key {key!r} has wrong type: " + f"expected {expected}, got {type(payload[key]).__name__}" + ) + # bool is a subclass of int in Python; reject `True`/`False` + # when the schema asked for an integer/number. + if expected in ("integer", "number") and isinstance(payload[key], bool): + return f"key {key!r} is a bool but schema asked for {expected}" + return None + + +def test_structured_outputs_anthropic(compat_result): + """Drive `claude --json-schema ...` against the LiteLLM proxy and + assert the trailing `result` event contains a schema-conforming + `structured_output`.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + outcomes = run_claude_models_parallel( + models=ANTHROPIC_MODELS, + prompt=PROMPT, + base_url=base_url, + api_key=api_key, + extra_args=["--json-schema", SCHEMA_JSON], + ) + + failures = [] + for model in ANTHROPIC_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + payload = _extract_structured_output(outcome.events) + if payload is None: + error = ( + f"[{model}] no `structured_output` in trailing result event; " + "Claude Code's StructuredOutput tool round-trip did not " + "complete end-to-end through the proxy" + ) + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + shape_error = _validate_against_schema(payload, SCHEMA) + if shape_error is not None: + error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/structured_outputs/test_azure.py b/tests/claude_code/structured_outputs/test_azure.py new file mode 100644 index 00000000000..0de4537fd08 --- /dev/null +++ b/tests/claude_code/structured_outputs/test_azure.py @@ -0,0 +1,220 @@ +"""structured_outputs x Azure (Microsoft Foundry). + +Drive the real `claude` CLI in headless mode with the `--json-schema` +flag, route through a LiteLLM proxy aimed at Azure (Microsoft Foundry), and assert that +the final stream-json `result` event surfaces a `structured_output` +object whose shape matches the schema. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/structured_outputs/test_azure.py + ^^^^^^^^^^^^^^^^^^ ^^^^^ + feature_id provider + +What this row actually exercises (and what it does not): + +`--json-schema` is implemented client-side by Claude Code: the CLI +synthesizes a synthetic `StructuredOutput` tool whose `input_schema` +equals the user-supplied JSON Schema, forces the model toward it, and +finally extracts the tool_use input on the trailing `result` event as +`structured_output: {...}`. The proxy never sees `output_config.schema` +in this flow -- it sees a normal `tools` array with one synthetic +tool. + +This makes the row a tool-use feature test in disguise. It's still a +distinct row from `tool_use` because: + + - The synthetic tool is generated per request from a user schema, not + a developer-declared one. Provider-side bugs that special-case + `Claude Code`-generated tool names (e.g. case-folding `tool_use` + blocks back to lowercase, or stripping the StructuredOutput-only + `additionalProperties: false`) only surface here. + - The success signal lives on the *final* `result` event, not the + intermediate `assistant` events the `tool_use` row checks. A proxy + that drops trailing events (seen in early Bedrock Converse SSE + plumbing) breaks this cell while leaving `tool_use` green. + +It is NOT a test of Anthropic's server-side `output_config.schema` +parameter -- that's a different feature used internally by Claude Code +for session-title generation and is not reachable from any CLI flag. +LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in +the `count_tokens` and other HTTP-probe rows, not here. + +Three Claude tiers run in parallel; one `compat_result.add(...)` per +tier so the matrix's "all three must pass" rule applies. +""" + +from __future__ import annotations + +import json +import os +import re +from typing import Any, Mapping, Optional, Sequence, Tuple + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +AZURE_MODELS = [ + "claude-haiku-4-5-azure", + "claude-sonnet-4-6-azure", + "claude-opus-4-7-azure", +] + +# Minimal schema with one required integer field. Kept intentionally +# small -- the matrix tests the *plumbing*, not the model's ability to +# satisfy a complex schema. A trivial arithmetic prompt + a one-field +# integer schema gives every tier (including Haiku) enough headroom +# that schema satisfaction is essentially deterministic, isolating +# failures to the proxy / transport. +SCHEMA = { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + "additionalProperties": False, +} +SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":")) + +# A prompt the model has no reason to misanswer; we don't check the +# value, but a wrong answer would suggest the structured-output +# pathway is silently degrading reasoning, which is itself worth +# noticing. +PROMPT = "What is 2 + 2? Reply only via the structured output." + + +def _extract_structured_output( + events: Sequence[Mapping[str, Any]], +) -> Optional[Mapping[str, Any]]: + """Return the `structured_output` payload from the last `result` event. + + Claude Code emits its terminal stream-json line as + `{"type":"result","structured_output":{...},...}` when a request + used `--json-schema` and the model actually produced a valid tool + call. If the model bailed or the proxy ate the trailing events, + `structured_output` is missing -- which is exactly the failure + mode we want this row to surface, so the caller treats `None` as + "feature did not work end-to-end". + """ + for event in reversed(list(events)): + if event.get("type") != "result": + continue + so = event.get("structured_output") + if isinstance(so, Mapping): + return so + return None + + +def _validate_against_schema( + payload: Mapping[str, Any], schema: Mapping[str, Any] +) -> Optional[str]: + """Tiny shape validator covering the subset we actually need. + + We deliberately do not pull in `jsonschema` as a test dep: the + matrix's success signal is "does the proxy let the synthetic + StructuredOutput tool round-trip end-to-end", and that's + answerable with a presence + type check over `required` keys. + Any malformed schema beyond that would be a Claude Code bug, + not a LiteLLM-proxy bug, so a deeper check would only add false + failures on the wrong axis. + """ + type_map = { + "integer": int, + "number": (int, float), + "string": str, + "boolean": bool, + "array": list, + "object": Mapping, + } + required = schema.get("required") or [] + properties = schema.get("properties") or {} + for key in required: + if key not in payload: + return f"missing required key {key!r}" + expected = (properties.get(key) or {}).get("type") + if expected and expected in type_map: + if not isinstance(payload[key], type_map[expected]): + return ( + f"key {key!r} has wrong type: " + f"expected {expected}, got {type(payload[key]).__name__}" + ) + # bool is a subclass of int in Python; reject `True`/`False` + # when the schema asked for an integer/number. + if expected in ("integer", "number") and isinstance(payload[key], bool): + return f"key {key!r} is a bool but schema asked for {expected}" + return None + + +def test_structured_outputs_azure(compat_result): + """Drive `claude --json-schema ...` against the LiteLLM proxy and + assert the trailing `result` event contains a schema-conforming + `structured_output`.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + outcomes = run_claude_models_parallel( + models=AZURE_MODELS, + prompt=PROMPT, + base_url=base_url, + api_key=api_key, + extra_args=["--json-schema", SCHEMA_JSON], + ) + + failures = [] + for model in AZURE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + payload = _extract_structured_output(outcome.events) + if payload is None: + error = ( + f"[{model}] no `structured_output` in trailing result event; " + "Claude Code's StructuredOutput tool round-trip did not " + "complete end-to-end through the proxy" + ) + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + shape_error = _validate_against_schema(payload, SCHEMA) + if shape_error is not None: + error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/structured_outputs/test_bedrock_converse.py b/tests/claude_code/structured_outputs/test_bedrock_converse.py new file mode 100644 index 00000000000..685e0c23838 --- /dev/null +++ b/tests/claude_code/structured_outputs/test_bedrock_converse.py @@ -0,0 +1,220 @@ +"""structured_outputs x Bedrock (Converse). + +Drive the real `claude` CLI in headless mode with the `--json-schema` +flag, route through a LiteLLM proxy aimed at Bedrock (Converse), and assert that +the final stream-json `result` event surfaces a `structured_output` +object whose shape matches the schema. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/structured_outputs/test_bedrock_converse.py + ^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider + +What this row actually exercises (and what it does not): + +`--json-schema` is implemented client-side by Claude Code: the CLI +synthesizes a synthetic `StructuredOutput` tool whose `input_schema` +equals the user-supplied JSON Schema, forces the model toward it, and +finally extracts the tool_use input on the trailing `result` event as +`structured_output: {...}`. The proxy never sees `output_config.schema` +in this flow -- it sees a normal `tools` array with one synthetic +tool. + +This makes the row a tool-use feature test in disguise. It's still a +distinct row from `tool_use` because: + + - The synthetic tool is generated per request from a user schema, not + a developer-declared one. Provider-side bugs that special-case + `Claude Code`-generated tool names (e.g. case-folding `tool_use` + blocks back to lowercase, or stripping the StructuredOutput-only + `additionalProperties: false`) only surface here. + - The success signal lives on the *final* `result` event, not the + intermediate `assistant` events the `tool_use` row checks. A proxy + that drops trailing events (seen in early Bedrock Converse SSE + plumbing) breaks this cell while leaving `tool_use` green. + +It is NOT a test of Anthropic's server-side `output_config.schema` +parameter -- that's a different feature used internally by Claude Code +for session-title generation and is not reachable from any CLI flag. +LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in +the `count_tokens` and other HTTP-probe rows, not here. + +Three Claude tiers run in parallel; one `compat_result.add(...)` per +tier so the matrix's "all three must pass" rule applies. +""" + +from __future__ import annotations + +import json +import os +import re +from typing import Any, Mapping, Optional, Sequence, Tuple + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_CONVERSE_MODELS = [ + "claude-haiku-4-5-bedrock-converse", + "claude-sonnet-4-6-bedrock-converse", + "claude-opus-4-7-bedrock-converse", +] + +# Minimal schema with one required integer field. Kept intentionally +# small -- the matrix tests the *plumbing*, not the model's ability to +# satisfy a complex schema. A trivial arithmetic prompt + a one-field +# integer schema gives every tier (including Haiku) enough headroom +# that schema satisfaction is essentially deterministic, isolating +# failures to the proxy / transport. +SCHEMA = { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + "additionalProperties": False, +} +SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":")) + +# A prompt the model has no reason to misanswer; we don't check the +# value, but a wrong answer would suggest the structured-output +# pathway is silently degrading reasoning, which is itself worth +# noticing. +PROMPT = "What is 2 + 2? Reply only via the structured output." + + +def _extract_structured_output( + events: Sequence[Mapping[str, Any]], +) -> Optional[Mapping[str, Any]]: + """Return the `structured_output` payload from the last `result` event. + + Claude Code emits its terminal stream-json line as + `{"type":"result","structured_output":{...},...}` when a request + used `--json-schema` and the model actually produced a valid tool + call. If the model bailed or the proxy ate the trailing events, + `structured_output` is missing -- which is exactly the failure + mode we want this row to surface, so the caller treats `None` as + "feature did not work end-to-end". + """ + for event in reversed(list(events)): + if event.get("type") != "result": + continue + so = event.get("structured_output") + if isinstance(so, Mapping): + return so + return None + + +def _validate_against_schema( + payload: Mapping[str, Any], schema: Mapping[str, Any] +) -> Optional[str]: + """Tiny shape validator covering the subset we actually need. + + We deliberately do not pull in `jsonschema` as a test dep: the + matrix's success signal is "does the proxy let the synthetic + StructuredOutput tool round-trip end-to-end", and that's + answerable with a presence + type check over `required` keys. + Any malformed schema beyond that would be a Claude Code bug, + not a LiteLLM-proxy bug, so a deeper check would only add false + failures on the wrong axis. + """ + type_map = { + "integer": int, + "number": (int, float), + "string": str, + "boolean": bool, + "array": list, + "object": Mapping, + } + required = schema.get("required") or [] + properties = schema.get("properties") or {} + for key in required: + if key not in payload: + return f"missing required key {key!r}" + expected = (properties.get(key) or {}).get("type") + if expected and expected in type_map: + if not isinstance(payload[key], type_map[expected]): + return ( + f"key {key!r} has wrong type: " + f"expected {expected}, got {type(payload[key]).__name__}" + ) + # bool is a subclass of int in Python; reject `True`/`False` + # when the schema asked for an integer/number. + if expected in ("integer", "number") and isinstance(payload[key], bool): + return f"key {key!r} is a bool but schema asked for {expected}" + return None + + +def test_structured_outputs_bedrock_converse(compat_result): + """Drive `claude --json-schema ...` against the LiteLLM proxy and + assert the trailing `result` event contains a schema-conforming + `structured_output`.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + outcomes = run_claude_models_parallel( + models=BEDROCK_CONVERSE_MODELS, + prompt=PROMPT, + base_url=base_url, + api_key=api_key, + extra_args=["--json-schema", SCHEMA_JSON], + ) + + failures = [] + for model in BEDROCK_CONVERSE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + payload = _extract_structured_output(outcome.events) + if payload is None: + error = ( + f"[{model}] no `structured_output` in trailing result event; " + "Claude Code's StructuredOutput tool round-trip did not " + "complete end-to-end through the proxy" + ) + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + shape_error = _validate_against_schema(payload, SCHEMA) + if shape_error is not None: + error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/structured_outputs/test_bedrock_invoke.py b/tests/claude_code/structured_outputs/test_bedrock_invoke.py new file mode 100644 index 00000000000..7fa014a4950 --- /dev/null +++ b/tests/claude_code/structured_outputs/test_bedrock_invoke.py @@ -0,0 +1,220 @@ +"""structured_outputs x Bedrock (Invoke). + +Drive the real `claude` CLI in headless mode with the `--json-schema` +flag, route through a LiteLLM proxy aimed at Bedrock (Invoke), and assert that +the final stream-json `result` event surfaces a `structured_output` +object whose shape matches the schema. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/structured_outputs/test_bedrock_invoke.py + ^^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider + +What this row actually exercises (and what it does not): + +`--json-schema` is implemented client-side by Claude Code: the CLI +synthesizes a synthetic `StructuredOutput` tool whose `input_schema` +equals the user-supplied JSON Schema, forces the model toward it, and +finally extracts the tool_use input on the trailing `result` event as +`structured_output: {...}`. The proxy never sees `output_config.schema` +in this flow -- it sees a normal `tools` array with one synthetic +tool. + +This makes the row a tool-use feature test in disguise. It's still a +distinct row from `tool_use` because: + + - The synthetic tool is generated per request from a user schema, not + a developer-declared one. Provider-side bugs that special-case + `Claude Code`-generated tool names (e.g. case-folding `tool_use` + blocks back to lowercase, or stripping the StructuredOutput-only + `additionalProperties: false`) only surface here. + - The success signal lives on the *final* `result` event, not the + intermediate `assistant` events the `tool_use` row checks. A proxy + that drops trailing events (seen in early Bedrock Converse SSE + plumbing) breaks this cell while leaving `tool_use` green. + +It is NOT a test of Anthropic's server-side `output_config.schema` +parameter -- that's a different feature used internally by Claude Code +for session-title generation and is not reachable from any CLI flag. +LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in +the `count_tokens` and other HTTP-probe rows, not here. + +Three Claude tiers run in parallel; one `compat_result.add(...)` per +tier so the matrix's "all three must pass" rule applies. +""" + +from __future__ import annotations + +import json +import os +import re +from typing import Any, Mapping, Optional, Sequence, Tuple + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_INVOKE_MODELS = [ + "claude-haiku-4-5-bedrock-invoke", + "claude-sonnet-4-6-bedrock-invoke", + "claude-opus-4-7-bedrock-invoke", +] + +# Minimal schema with one required integer field. Kept intentionally +# small -- the matrix tests the *plumbing*, not the model's ability to +# satisfy a complex schema. A trivial arithmetic prompt + a one-field +# integer schema gives every tier (including Haiku) enough headroom +# that schema satisfaction is essentially deterministic, isolating +# failures to the proxy / transport. +SCHEMA = { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + "additionalProperties": False, +} +SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":")) + +# A prompt the model has no reason to misanswer; we don't check the +# value, but a wrong answer would suggest the structured-output +# pathway is silently degrading reasoning, which is itself worth +# noticing. +PROMPT = "What is 2 + 2? Reply only via the structured output." + + +def _extract_structured_output( + events: Sequence[Mapping[str, Any]], +) -> Optional[Mapping[str, Any]]: + """Return the `structured_output` payload from the last `result` event. + + Claude Code emits its terminal stream-json line as + `{"type":"result","structured_output":{...},...}` when a request + used `--json-schema` and the model actually produced a valid tool + call. If the model bailed or the proxy ate the trailing events, + `structured_output` is missing -- which is exactly the failure + mode we want this row to surface, so the caller treats `None` as + "feature did not work end-to-end". + """ + for event in reversed(list(events)): + if event.get("type") != "result": + continue + so = event.get("structured_output") + if isinstance(so, Mapping): + return so + return None + + +def _validate_against_schema( + payload: Mapping[str, Any], schema: Mapping[str, Any] +) -> Optional[str]: + """Tiny shape validator covering the subset we actually need. + + We deliberately do not pull in `jsonschema` as a test dep: the + matrix's success signal is "does the proxy let the synthetic + StructuredOutput tool round-trip end-to-end", and that's + answerable with a presence + type check over `required` keys. + Any malformed schema beyond that would be a Claude Code bug, + not a LiteLLM-proxy bug, so a deeper check would only add false + failures on the wrong axis. + """ + type_map = { + "integer": int, + "number": (int, float), + "string": str, + "boolean": bool, + "array": list, + "object": Mapping, + } + required = schema.get("required") or [] + properties = schema.get("properties") or {} + for key in required: + if key not in payload: + return f"missing required key {key!r}" + expected = (properties.get(key) or {}).get("type") + if expected and expected in type_map: + if not isinstance(payload[key], type_map[expected]): + return ( + f"key {key!r} has wrong type: " + f"expected {expected}, got {type(payload[key]).__name__}" + ) + # bool is a subclass of int in Python; reject `True`/`False` + # when the schema asked for an integer/number. + if expected in ("integer", "number") and isinstance(payload[key], bool): + return f"key {key!r} is a bool but schema asked for {expected}" + return None + + +def test_structured_outputs_bedrock_invoke(compat_result): + """Drive `claude --json-schema ...` against the LiteLLM proxy and + assert the trailing `result` event contains a schema-conforming + `structured_output`.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + outcomes = run_claude_models_parallel( + models=BEDROCK_INVOKE_MODELS, + prompt=PROMPT, + base_url=base_url, + api_key=api_key, + extra_args=["--json-schema", SCHEMA_JSON], + ) + + failures = [] + for model in BEDROCK_INVOKE_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + payload = _extract_structured_output(outcome.events) + if payload is None: + error = ( + f"[{model}] no `structured_output` in trailing result event; " + "Claude Code's StructuredOutput tool round-trip did not " + "complete end-to-end through the proxy" + ) + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + shape_error = _validate_against_schema(payload, SCHEMA) + if shape_error is not None: + error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/structured_outputs/test_vertex_ai.py b/tests/claude_code/structured_outputs/test_vertex_ai.py new file mode 100644 index 00000000000..1bfadfdb439 --- /dev/null +++ b/tests/claude_code/structured_outputs/test_vertex_ai.py @@ -0,0 +1,220 @@ +"""structured_outputs x Vertex AI. + +Drive the real `claude` CLI in headless mode with the `--json-schema` +flag, route through a LiteLLM proxy aimed at Vertex AI, and assert that +the final stream-json `result` event surfaces a `structured_output` +object whose shape matches the schema. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/structured_outputs/test_vertex_ai.py + ^^^^^^^^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +What this row actually exercises (and what it does not): + +`--json-schema` is implemented client-side by Claude Code: the CLI +synthesizes a synthetic `StructuredOutput` tool whose `input_schema` +equals the user-supplied JSON Schema, forces the model toward it, and +finally extracts the tool_use input on the trailing `result` event as +`structured_output: {...}`. The proxy never sees `output_config.schema` +in this flow -- it sees a normal `tools` array with one synthetic +tool. + +This makes the row a tool-use feature test in disguise. It's still a +distinct row from `tool_use` because: + + - The synthetic tool is generated per request from a user schema, not + a developer-declared one. Provider-side bugs that special-case + `Claude Code`-generated tool names (e.g. case-folding `tool_use` + blocks back to lowercase, or stripping the StructuredOutput-only + `additionalProperties: false`) only surface here. + - The success signal lives on the *final* `result` event, not the + intermediate `assistant` events the `tool_use` row checks. A proxy + that drops trailing events (seen in early Bedrock Converse SSE + plumbing) breaks this cell while leaving `tool_use` green. + +It is NOT a test of Anthropic's server-side `output_config.schema` +parameter -- that's a different feature used internally by Claude Code +for session-title generation and is not reachable from any CLI flag. +LiteLLM's `output_config`-stripping fixes (2.1.122, 2.1.81) surface in +the `count_tokens` and other HTTP-probe rows, not here. + +Three Claude tiers run in parallel; one `compat_result.add(...)` per +tier so the matrix's "all three must pass" rule applies. +""" + +from __future__ import annotations + +import json +import os +import re +from typing import Any, Mapping, Optional, Sequence, Tuple + +import pytest + +from tests.claude_code.cli_driver import ( + ClaudeCLIError, + failure_diagnostic, + run_claude_models_parallel, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +VERTEX_AI_MODELS = [ + "claude-haiku-4-5-vertex", + "claude-sonnet-4-6-vertex", + "claude-opus-4-7-vertex", +] + +# Minimal schema with one required integer field. Kept intentionally +# small -- the matrix tests the *plumbing*, not the model's ability to +# satisfy a complex schema. A trivial arithmetic prompt + a one-field +# integer schema gives every tier (including Haiku) enough headroom +# that schema satisfaction is essentially deterministic, isolating +# failures to the proxy / transport. +SCHEMA = { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + "additionalProperties": False, +} +SCHEMA_JSON = json.dumps(SCHEMA, separators=(",", ":")) + +# A prompt the model has no reason to misanswer; we don't check the +# value, but a wrong answer would suggest the structured-output +# pathway is silently degrading reasoning, which is itself worth +# noticing. +PROMPT = "What is 2 + 2? Reply only via the structured output." + + +def _extract_structured_output( + events: Sequence[Mapping[str, Any]], +) -> Optional[Mapping[str, Any]]: + """Return the `structured_output` payload from the last `result` event. + + Claude Code emits its terminal stream-json line as + `{"type":"result","structured_output":{...},...}` when a request + used `--json-schema` and the model actually produced a valid tool + call. If the model bailed or the proxy ate the trailing events, + `structured_output` is missing -- which is exactly the failure + mode we want this row to surface, so the caller treats `None` as + "feature did not work end-to-end". + """ + for event in reversed(list(events)): + if event.get("type") != "result": + continue + so = event.get("structured_output") + if isinstance(so, Mapping): + return so + return None + + +def _validate_against_schema( + payload: Mapping[str, Any], schema: Mapping[str, Any] +) -> Optional[str]: + """Tiny shape validator covering the subset we actually need. + + We deliberately do not pull in `jsonschema` as a test dep: the + matrix's success signal is "does the proxy let the synthetic + StructuredOutput tool round-trip end-to-end", and that's + answerable with a presence + type check over `required` keys. + Any malformed schema beyond that would be a Claude Code bug, + not a LiteLLM-proxy bug, so a deeper check would only add false + failures on the wrong axis. + """ + type_map = { + "integer": int, + "number": (int, float), + "string": str, + "boolean": bool, + "array": list, + "object": Mapping, + } + required = schema.get("required") or [] + properties = schema.get("properties") or {} + for key in required: + if key not in payload: + return f"missing required key {key!r}" + expected = (properties.get(key) or {}).get("type") + if expected and expected in type_map: + if not isinstance(payload[key], type_map[expected]): + return ( + f"key {key!r} has wrong type: " + f"expected {expected}, got {type(payload[key]).__name__}" + ) + # bool is a subclass of int in Python; reject `True`/`False` + # when the schema asked for an integer/number. + if expected in ("integer", "number") and isinstance(payload[key], bool): + return f"key {key!r} is a bool but schema asked for {expected}" + return None + + +def test_structured_outputs_vertex_ai(compat_result): + """Drive `claude --json-schema ...` against the LiteLLM proxy and + assert the trailing `result` event contains a schema-conforming + `structured_output`.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + outcomes = run_claude_models_parallel( + models=VERTEX_AI_MODELS, + prompt=PROMPT, + base_url=base_url, + api_key=api_key, + extra_args=["--json-schema", SCHEMA_JSON], + ) + + failures = [] + for model in VERTEX_AI_MODELS: + outcome = outcomes[model] + if isinstance(outcome, ClaudeCLIError): + error = f"[{model}] {outcome}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + if outcome.exit_code != 0: + error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + payload = _extract_structured_output(outcome.events) + if payload is None: + error = ( + f"[{model}] no `structured_output` in trailing result event; " + "Claude Code's StructuredOutput tool round-trip did not " + "complete end-to-end through the proxy" + ) + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + shape_error = _validate_against_schema(payload, SCHEMA) + if shape_error is not None: + error = f"[{model}] structured_output shape error: {shape_error}; payload={payload}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/thinking/__init__.py b/tests/claude_code/thinking/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/extended_thinking/test_anthropic.py b/tests/claude_code/thinking/test_anthropic.py similarity index 94% rename from tests/claude_code/extended_thinking/test_anthropic.py rename to tests/claude_code/thinking/test_anthropic.py index 9dbe7a354bb..0d887957b75 100644 --- a/tests/claude_code/extended_thinking/test_anthropic.py +++ b/tests/claude_code/thinking/test_anthropic.py @@ -1,4 +1,4 @@ -"""extended_thinking x Anthropic. +"""thinking x Anthropic. Drive the real `claude` CLI against a running LiteLLM proxy that routes to Anthropic, enable extended thinking via `--effort high`, and assert @@ -9,9 +9,9 @@ upstream response's `thinking` content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: - tests/claude_code/extended_thinking/test_anthropic.py - ^^^^^^^^^^^^^^^^^ ^^^^^^^^^ - feature_id provider + tests/claude_code/thinking/test_anthropic.py + ^^^^^^^^ ^^^^^^^^^ + feature_id provider The three Claude tiers run in parallel inside this single test, with one `compat_result.add(...)` entry per model so the matrix builder @@ -76,7 +76,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: return False -def test_extended_thinking_anthropic(compat_result): +def test_thinking_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) diff --git a/tests/claude_code/extended_thinking/test_azure.py b/tests/claude_code/thinking/test_azure.py similarity index 93% rename from tests/claude_code/extended_thinking/test_azure.py rename to tests/claude_code/thinking/test_azure.py index 3f243cd3dd4..bf9d93f8304 100644 --- a/tests/claude_code/extended_thinking/test_azure.py +++ b/tests/claude_code/thinking/test_azure.py @@ -1,4 +1,4 @@ -"""extended_thinking x Azure (Microsoft Foundry). +"""thinking x Azure (Microsoft Foundry). Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Anthropic's models hosted in Microsoft Foundry on @@ -16,9 +16,9 @@ re-evaluate. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: - tests/claude_code/extended_thinking/test_azure.py - ^^^^^^^^^^^^^^^^^ ^^^^^ - feature_id provider + tests/claude_code/thinking/test_azure.py + ^^^^^^^^ ^^^^^ + feature_id provider """ from __future__ import annotations @@ -64,7 +64,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: return False -def test_extended_thinking_azure(compat_result): +def test_thinking_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) diff --git a/tests/claude_code/extended_thinking/test_bedrock_converse.py b/tests/claude_code/thinking/test_bedrock_converse.py similarity index 92% rename from tests/claude_code/extended_thinking/test_bedrock_converse.py rename to tests/claude_code/thinking/test_bedrock_converse.py index 2ccb3a44729..f55b0cbe01a 100644 --- a/tests/claude_code/extended_thinking/test_bedrock_converse.py +++ b/tests/claude_code/thinking/test_bedrock_converse.py @@ -1,4 +1,4 @@ -"""extended_thinking x Bedrock (Converse). +"""thinking x Bedrock (Converse). Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the unified `Converse` API path, @@ -8,9 +8,9 @@ upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: - tests/claude_code/extended_thinking/test_bedrock_converse.py - ^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ - feature_id provider + tests/claude_code/thinking/test_bedrock_converse.py + ^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider """ from __future__ import annotations @@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: return False -def test_extended_thinking_bedrock_converse(compat_result): +def test_thinking_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) diff --git a/tests/claude_code/extended_thinking/test_bedrock_invoke.py b/tests/claude_code/thinking/test_bedrock_invoke.py similarity index 92% rename from tests/claude_code/extended_thinking/test_bedrock_invoke.py rename to tests/claude_code/thinking/test_bedrock_invoke.py index 79557d40a91..8d977a4e716 100644 --- a/tests/claude_code/extended_thinking/test_bedrock_invoke.py +++ b/tests/claude_code/thinking/test_bedrock_invoke.py @@ -1,4 +1,4 @@ -"""extended_thinking x Bedrock (Invoke). +"""thinking x Bedrock (Invoke). Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the legacy `InvokeModel` API path, @@ -8,9 +8,9 @@ upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: - tests/claude_code/extended_thinking/test_bedrock_invoke.py - ^^^^^^^^^^^^^^^^^ ^^^^^^^^^^^^^^ - feature_id provider + tests/claude_code/thinking/test_bedrock_invoke.py + ^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider """ from __future__ import annotations @@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: return False -def test_extended_thinking_bedrock_invoke(compat_result): +def test_thinking_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) diff --git a/tests/claude_code/extended_thinking/test_vertex_ai.py b/tests/claude_code/thinking/test_vertex_ai.py similarity index 92% rename from tests/claude_code/extended_thinking/test_vertex_ai.py rename to tests/claude_code/thinking/test_vertex_ai.py index 7b37866451e..648db6bc835 100644 --- a/tests/claude_code/extended_thinking/test_vertex_ai.py +++ b/tests/claude_code/thinking/test_vertex_ai.py @@ -1,4 +1,4 @@ -"""extended_thinking x Vertex AI. +"""thinking x Vertex AI. Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Anthropic's models on Google Cloud Vertex AI, enable @@ -8,9 +8,9 @@ upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: - tests/claude_code/extended_thinking/test_vertex_ai.py - ^^^^^^^^^^^^^^^^^ ^^^^^^^^^ - feature_id provider + tests/claude_code/thinking/test_vertex_ai.py + ^^^^^^^^ ^^^^^^^^^ + feature_id provider """ from __future__ import annotations @@ -56,7 +56,7 @@ def _has_thinking_block(events: Sequence[Mapping[str, Any]]) -> bool: return False -def test_extended_thinking_vertex_ai(compat_result): +def test_thinking_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with thinking enabled and assert a `thinking` content block was emitted.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) diff --git a/tests/claude_code/tool_search/__init__.py b/tests/claude_code/tool_search/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/claude_code/tool_search/test_anthropic.py b/tests/claude_code/tool_search/test_anthropic.py new file mode 100644 index 00000000000..cffec93419b --- /dev/null +++ b/tests/claude_code/tool_search/test_anthropic.py @@ -0,0 +1,99 @@ +"""tool_search x Anthropic. + +HTTP-probe row. Sends a single `/v1/messages` request whose `tools` +array includes a `tool_search_tool_regex_20251119` discovery tool, and +asserts the proxy round-trips it to the upstream without a 400. This +verifies LiteLLM's tool-search beta-header translation +(`advanced-tool-use-2025-11-20` for Anthropic-shape providers, +`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/tool_search/test_anthropic.py + ^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI / MCP fan-out: + +Real Claude Code activates tool_search by registering >N MCP tools and +relying on the model's internal heuristic to call the discovery tool +before any user tool. That setup requires standing up a stub MCP +server that exposes 50+ tool stubs and depends on Claude Code's +auto-deferral heuristic continuing to fire at today's tool count -- +both of which break silently when Claude Code's threshold changes +between releases. + +The bugs LiteLLM has actually shipped fixes for in this area +(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are +beta-header translation and proxy-side type recognition, not MCP +fan-out behavior. An HTTP probe hits exactly that surface: the +request goes out with a `tool_search_tool_regex_20251119` tool type, +the proxy is responsible for attaching the per-provider beta header +and forwarding, and the upstream either accepts or 400s. A red cell +here is always a proxy-side regression, not a flaky model-behavior +artifact. + +Three Claude tiers are probed in sequence (count is too low to be +worth the parallelism overhead, and HTTP probes don't compete for +the proxy's `--num-workers` slots the way CLI subprocess runs do). +The matrix's "all three must pass" rule still applies via the +per-cell aggregator. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_tool_search_shape, + probe_tool_search, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +ANTHROPIC_MODELS = [ + "claude-haiku-4-5", + "claude-sonnet-4-6", + "claude-opus-4-7", +] + + +def test_tool_search_anthropic(compat_result): + """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` + tool and assert the proxy + upstream accept it for every Anthropic + tier.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in ANTHROPIC_MODELS: + result = probe_tool_search(base_url=base_url, api_key=api_key, model=model) + shape_error = assert_tool_search_shape(result) + if shape_error is not None: + error = f"[{model}] tool_search probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/tool_search/test_azure.py b/tests/claude_code/tool_search/test_azure.py new file mode 100644 index 00000000000..992494abb71 --- /dev/null +++ b/tests/claude_code/tool_search/test_azure.py @@ -0,0 +1,99 @@ +"""tool_search x Azure (Microsoft Foundry). + +HTTP-probe row. Sends a single `/v1/messages` request whose `tools` +array includes a `tool_search_tool_regex_20251119` discovery tool, and +asserts the proxy round-trips it to the upstream without a 400. This +verifies LiteLLM's tool-search beta-header translation +(`advanced-tool-use-2025-11-20` for Anthropic-shape providers, +`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/tool_search/test_azure.py + ^^^^^^^^^^^ ^^^^^ + feature_id provider + +Why HTTP probe instead of CLI / MCP fan-out: + +Real Claude Code activates tool_search by registering >N MCP tools and +relying on the model's internal heuristic to call the discovery tool +before any user tool. That setup requires standing up a stub MCP +server that exposes 50+ tool stubs and depends on Claude Code's +auto-deferral heuristic continuing to fire at today's tool count -- +both of which break silently when Claude Code's threshold changes +between releases. + +The bugs LiteLLM has actually shipped fixes for in this area +(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are +beta-header translation and proxy-side type recognition, not MCP +fan-out behavior. An HTTP probe hits exactly that surface: the +request goes out with a `tool_search_tool_regex_20251119` tool type, +the proxy is responsible for attaching the per-provider beta header +and forwarding, and the upstream either accepts or 400s. A red cell +here is always a proxy-side regression, not a flaky model-behavior +artifact. + +Three Claude tiers are probed in sequence (count is too low to be +worth the parallelism overhead, and HTTP probes don't compete for +the proxy's `--num-workers` slots the way CLI subprocess runs do). +The matrix's "all three must pass" rule still applies via the +per-cell aggregator. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_tool_search_shape, + probe_tool_search, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +AZURE_MODELS = [ + "claude-haiku-4-5-azure", + "claude-sonnet-4-6-azure", + "claude-opus-4-7-azure", +] + + +def test_tool_search_azure(compat_result): + """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` + tool and assert the proxy + upstream accept it for every Azure (Microsoft Foundry) + tier.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in AZURE_MODELS: + result = probe_tool_search(base_url=base_url, api_key=api_key, model=model) + shape_error = assert_tool_search_shape(result) + if shape_error is not None: + error = f"[{model}] tool_search probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/tool_search/test_bedrock_converse.py b/tests/claude_code/tool_search/test_bedrock_converse.py new file mode 100644 index 00000000000..49a3d12b4d3 --- /dev/null +++ b/tests/claude_code/tool_search/test_bedrock_converse.py @@ -0,0 +1,99 @@ +"""tool_search x Bedrock (Converse). + +HTTP-probe row. Sends a single `/v1/messages` request whose `tools` +array includes a `tool_search_tool_regex_20251119` discovery tool, and +asserts the proxy round-trips it to the upstream without a 400. This +verifies LiteLLM's tool-search beta-header translation +(`advanced-tool-use-2025-11-20` for Anthropic-shape providers, +`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/tool_search/test_bedrock_converse.py + ^^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI / MCP fan-out: + +Real Claude Code activates tool_search by registering >N MCP tools and +relying on the model's internal heuristic to call the discovery tool +before any user tool. That setup requires standing up a stub MCP +server that exposes 50+ tool stubs and depends on Claude Code's +auto-deferral heuristic continuing to fire at today's tool count -- +both of which break silently when Claude Code's threshold changes +between releases. + +The bugs LiteLLM has actually shipped fixes for in this area +(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are +beta-header translation and proxy-side type recognition, not MCP +fan-out behavior. An HTTP probe hits exactly that surface: the +request goes out with a `tool_search_tool_regex_20251119` tool type, +the proxy is responsible for attaching the per-provider beta header +and forwarding, and the upstream either accepts or 400s. A red cell +here is always a proxy-side regression, not a flaky model-behavior +artifact. + +Three Claude tiers are probed in sequence (count is too low to be +worth the parallelism overhead, and HTTP probes don't compete for +the proxy's `--num-workers` slots the way CLI subprocess runs do). +The matrix's "all three must pass" rule still applies via the +per-cell aggregator. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_tool_search_shape, + probe_tool_search, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_CONVERSE_MODELS = [ + "claude-haiku-4-5-bedrock-converse", + "claude-sonnet-4-6-bedrock-converse", + "claude-opus-4-7-bedrock-converse", +] + + +def test_tool_search_bedrock_converse(compat_result): + """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` + tool and assert the proxy + upstream accept it for every Bedrock (Converse) + tier.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in BEDROCK_CONVERSE_MODELS: + result = probe_tool_search(base_url=base_url, api_key=api_key, model=model) + shape_error = assert_tool_search_shape(result) + if shape_error is not None: + error = f"[{model}] tool_search probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/tool_search/test_bedrock_invoke.py b/tests/claude_code/tool_search/test_bedrock_invoke.py new file mode 100644 index 00000000000..6d7c9ed47ad --- /dev/null +++ b/tests/claude_code/tool_search/test_bedrock_invoke.py @@ -0,0 +1,99 @@ +"""tool_search x Bedrock (Invoke). + +HTTP-probe row. Sends a single `/v1/messages` request whose `tools` +array includes a `tool_search_tool_regex_20251119` discovery tool, and +asserts the proxy round-trips it to the upstream without a 400. This +verifies LiteLLM's tool-search beta-header translation +(`advanced-tool-use-2025-11-20` for Anthropic-shape providers, +`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/tool_search/test_bedrock_invoke.py + ^^^^^^^^^^^ ^^^^^^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI / MCP fan-out: + +Real Claude Code activates tool_search by registering >N MCP tools and +relying on the model's internal heuristic to call the discovery tool +before any user tool. That setup requires standing up a stub MCP +server that exposes 50+ tool stubs and depends on Claude Code's +auto-deferral heuristic continuing to fire at today's tool count -- +both of which break silently when Claude Code's threshold changes +between releases. + +The bugs LiteLLM has actually shipped fixes for in this area +(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are +beta-header translation and proxy-side type recognition, not MCP +fan-out behavior. An HTTP probe hits exactly that surface: the +request goes out with a `tool_search_tool_regex_20251119` tool type, +the proxy is responsible for attaching the per-provider beta header +and forwarding, and the upstream either accepts or 400s. A red cell +here is always a proxy-side regression, not a flaky model-behavior +artifact. + +Three Claude tiers are probed in sequence (count is too low to be +worth the parallelism overhead, and HTTP probes don't compete for +the proxy's `--num-workers` slots the way CLI subprocess runs do). +The matrix's "all three must pass" rule still applies via the +per-cell aggregator. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_tool_search_shape, + probe_tool_search, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +BEDROCK_INVOKE_MODELS = [ + "claude-haiku-4-5-bedrock-invoke", + "claude-sonnet-4-6-bedrock-invoke", + "claude-opus-4-7-bedrock-invoke", +] + + +def test_tool_search_bedrock_invoke(compat_result): + """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` + tool and assert the proxy + upstream accept it for every Bedrock (Invoke) + tier.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in BEDROCK_INVOKE_MODELS: + result = probe_tool_search(base_url=base_url, api_key=api_key, model=model) + shape_error = assert_tool_search_shape(result) + if shape_error is not None: + error = f"[{model}] tool_search probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False) diff --git a/tests/claude_code/tool_search/test_vertex_ai.py b/tests/claude_code/tool_search/test_vertex_ai.py new file mode 100644 index 00000000000..c934cde211c --- /dev/null +++ b/tests/claude_code/tool_search/test_vertex_ai.py @@ -0,0 +1,99 @@ +"""tool_search x Vertex AI. + +HTTP-probe row. Sends a single `/v1/messages` request whose `tools` +array includes a `tool_search_tool_regex_20251119` discovery tool, and +asserts the proxy round-trips it to the upstream without a 400. This +verifies LiteLLM's tool-search beta-header translation +(`advanced-tool-use-2025-11-20` for Anthropic-shape providers, +`tool-search-tool-2025-10-19` for Vertex/Bedrock) survives end-to-end. + +The (feature, provider) for this cell is inferred from the file path by +`tests/claude_code/conftest.py`: + + tests/claude_code/tool_search/test_vertex_ai.py + ^^^^^^^^^^^ ^^^^^^^^^ + feature_id provider + +Why HTTP probe instead of CLI / MCP fan-out: + +Real Claude Code activates tool_search by registering >N MCP tools and +relying on the model's internal heuristic to call the discovery tool +before any user tool. That setup requires standing up a stub MCP +server that exposes 50+ tool stubs and depends on Claude Code's +auto-deferral heuristic continuing to fire at today's tool count -- +both of which break silently when Claude Code's threshold changes +between releases. + +The bugs LiteLLM has actually shipped fixes for in this area +(2.1.117, 2.1.72, 2.1.70 in the Claude Code release notes) are +beta-header translation and proxy-side type recognition, not MCP +fan-out behavior. An HTTP probe hits exactly that surface: the +request goes out with a `tool_search_tool_regex_20251119` tool type, +the proxy is responsible for attaching the per-provider beta header +and forwarding, and the upstream either accepts or 400s. A red cell +here is always a proxy-side regression, not a flaky model-behavior +artifact. + +Three Claude tiers are probed in sequence (count is too low to be +worth the parallelism overhead, and HTTP probes don't compete for +the proxy's `--num-workers` slots the way CLI subprocess runs do). +The matrix's "all three must pass" rule still applies via the +per-cell aggregator. +""" + +from __future__ import annotations + +import os + +import pytest + +from tests.claude_code.http_probe import ( + assert_tool_search_shape, + probe_tool_search, +) + +PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL" +PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY" + +VERTEX_AI_MODELS = [ + "claude-haiku-4-5-vertex", + "claude-sonnet-4-6-vertex", + "claude-opus-4-7-vertex", +] + + +def test_tool_search_vertex_ai(compat_result): + """Probe `/v1/messages` with a `tool_search_tool_regex_20251119` + tool and assert the proxy + upstream accept it for every Vertex AI + tier.""" + base_url = os.environ.get(PROXY_BASE_URL_ENV) + api_key = os.environ.get(PROXY_API_KEY_ENV) + if not base_url or not api_key: + compat_result.set( + { + "status": "fail", + "error": ( + f"missing required env: set {PROXY_BASE_URL_ENV} and " + f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy" + ), + } + ) + pytest.fail( + f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", + pytrace=False, + ) + + failures = [] + for model in VERTEX_AI_MODELS: + result = probe_tool_search(base_url=base_url, api_key=api_key, model=model) + shape_error = assert_tool_search_shape(result) + if shape_error is not None: + error = f"[{model}] tool_search probe failed: {shape_error}" + compat_result.add({"status": "fail", "error": error}) + failures.append(error) + continue + + compat_result.add({"status": "pass"}) + + if failures: + pytest.fail("; ".join(failures), pytrace=False)