compat-matrix: fix vision, extended_thinking, web_search test bugs

These three cells were failing for reasons unrelated to LiteLLM
translation:

- vision: tests passed `--image <path>`, a flag that no longer exists
  in Claude Code 2.x (image attachment is now via the Files API or via
  `--input-format stream-json` with inline content blocks). Rewrite
  the cells to feed an Anthropic-shaped user message containing both
  text and a base64 `image` content block through stdin in stream-json
  mode. Hermetic — no temp file or Files API upload needed.

- extended_thinking: tests set `MAX_THINKING_TOKENS=4096` as an env
  var, which Claude Code 2.x ignores. Switch to `--effort max` (the
  current CLI knob) and use a non-trivial prompt (3-gallon / 5-gallon
  jug puzzle). With trivial arithmetic the modern Sonnet/Opus tiers
  optimize away the thinking step and arrive without a thinking block,
  which made the test silently false-fail.

- web_search: assertion looked for `server_tool_use` /
  `web_search_tool_result` blocks, but Claude Code's `WebSearch` is
  a *client-side* tool: the CLI executes the search itself and feeds
  the result back as a regular `tool_result` block. The Anthropic
  server-side `web_search_20250305` tool only fires when injected
  into the request directly (which the CLI does not do). Update the
  assertion to look for a `tool_use` block whose name is
  `WebSearch` — that's the right signal that the proxy preserved
  both the request-side tool definition and the response-side tool_use
  block end-to-end.

Driver change required to support stream-json input + variadic flags:

- cli_driver: insert `--` before the prompt positional. Variadic
  flags like `--allowed-tools <tools...>` (commander.js) greedily
  consume every following token, so the prompt was being eaten as a
  tool name and the CLI would error out with "Input must be provided
  either through stdin or as a prompt argument when using --print".
- cli_driver: thread a `stdin_input` parameter through `run_claude`
  and `run_claude_models_parallel` so the vision rewrite can pipe
  stream-json events to the CLI on stdin.

Validated end-to-end against a live LiteLLM proxy: all three Anthropic
cells now pass on Haiku 4.5, Sonnet 4.6, and Opus 4.7. Driver unit
tests (120) still green.
This commit is contained in:
mateo-berri 2026-05-07 02:16:10 +00:00
parent 4caf8a6d37
commit 127bbd2b4c
22 changed files with 603 additions and 225 deletions

View file

@ -34,10 +34,11 @@ class _Completed:
def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
captured = {}
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
captured["cmd"] = cmd
captured["env"] = env
captured["timeout"] = timeout
captured["input"] = input
return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
return runner, captured
@ -61,16 +62,19 @@ def test_run_claude_assembles_command_correctly():
assert "stream-json" in cmd
assert "--model" in cmd
assert "claude-haiku-4-5" in cmd
assert cmd[-1] == "hello"
# prompt is the last positional after the `--` end-of-options marker.
assert cmd[-2:] == ["--", "hello"]
def test_run_claude_places_extra_args_before_prompt():
"""`claude --print` expects the prompt as the final positional arg.
Flags appearing after the prompt are ignored or eaten by the prompt
parser, which silently broke the tool_use and vision cells before the
fix. Pin the ordering: every flag (including caller-supplied
`extra_args`) must precede the prompt.
parser (especially variadic flags like `--allowed-tools <tools...>`),
which silently broke the tool_use, vision, and web_search cells
before the fix. Pin the ordering: every flag (including
caller-supplied `extra_args`) must precede the `--` end-of-options
marker, which itself precedes the prompt.
"""
runner, captured = _make_runner(stdout="")
run_claude(
@ -82,9 +86,15 @@ def test_run_claude_places_extra_args_before_prompt():
runner=runner,
)
cmd = captured["cmd"]
assert cmd[-1] == "say hi"
prompt_idx = cmd.index("say hi")
assert cmd[prompt_idx - 2 : prompt_idx] == ["--allowed-tools", "Bash"]
assert cmd[-3:] == ["--allowed-tools", "Bash", "--"] or cmd[-2:] == [
"--",
"say hi",
]
# Stronger: prompt is last, `--` immediately precedes it, and the
# caller's extra_args sit somewhere earlier in the command.
assert cmd[-2:] == ["--", "say hi"]
assert "--allowed-tools" in cmd
assert cmd.index("--allowed-tools") < cmd.index("--")
def test_run_claude_overlays_proxy_env():
@ -414,7 +424,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model():
"""Each model gets its own DriverResult keyed under the helper's dict."""
seen_models: List[str] = []
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
# The model id is two slots after `--model` in the assembled command.
idx = cmd.index("--model")
model = cmd[idx + 1]
@ -452,7 +462,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model():
def test_run_claude_models_parallel_returns_errors_as_values():
"""A model whose CLI is missing surfaces as a ClaudeCLIError, not a raise."""
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
if model == "boom":
@ -485,7 +495,7 @@ def test_run_claude_models_parallel_returns_errors_as_values():
def test_run_claude_models_parallel_preserves_nonzero_exit_codes():
"""Mixed success/failure on exit code should not collapse into one verdict."""
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
if model == "fail":
@ -536,7 +546,7 @@ def test_run_claude_models_parallel_stamps_duration_on_each_result():
"""
import time
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
model = cmd[idx + 1]
time.sleep(0.05 if model == "fast" else 0.40)
@ -565,7 +575,7 @@ def test_run_claude_models_parallel_breakdown_logs_to_stderr(capsys):
"""The breakdown helper must emit a per-model timing block so users
can answer "why didn't parallel help?" without re-instrumenting."""
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
return _Completed(returncode=0, stdout="")
run_claude_models_parallel(
@ -588,7 +598,7 @@ def test_run_claude_models_parallel_breakdown_marks_cli_errors(capsys):
"""When a model raises ClaudeCLIError, the breakdown should still
show its row tagged as `cli-error` rather than crashing or omitting it."""
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
idx = cmd.index("--model")
if cmd[idx + 1] == "boom":
raise FileNotFoundError(2, "no such file", "claude")
@ -613,7 +623,7 @@ def test_run_claude_models_parallel_forwards_extra_args_and_env():
captured_envs: List[dict] = []
captured_cmds: List[List[str]] = []
def runner(cmd, env, capture_output, text, timeout, check):
def runner(cmd, env, capture_output, text, timeout, check, input=None):
captured_envs.append(env)
captured_cmds.append(cmd)
return _Completed(returncode=0, stdout="")

View file

@ -90,12 +90,13 @@ class DriverResult:
def run_claude(
*,
prompt: str,
prompt: Optional[str],
model: str,
base_url: str,
api_key: str,
extra_env: Optional[Mapping[str, str]] = None,
extra_args: Optional[Sequence[str]] = None,
stdin_input: Optional[str] = None,
cli_path: str = CLAUDE_CLI_DEFAULT,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
runner: Optional[Any] = None,
@ -118,8 +119,10 @@ def run_claude(
process-wide singleton; unit tests pass a no-op limiter or one
backed by a tmp dir to keep tests hermetic.
"""
if not prompt:
raise ValueError("prompt must be a non-empty string")
if prompt is None and stdin_input is None:
raise ValueError("must supply either `prompt` or `stdin_input`")
if prompt is not None and not prompt:
raise ValueError("prompt must be a non-empty string when provided")
if not model:
raise ValueError("model must be a non-empty string")
if not base_url:
@ -132,6 +135,14 @@ def run_claude(
# prompt (or silently dropped, depending on the CLI version) and the
# tool_use / vision cells fail with confusing "no tool_use observed"
# errors. Build the flag list first, then append the prompt last.
#
# When `extra_args` contains a *variadic* flag like `--allowed-tools
# WebSearch` (commander.js's `<tools...>`), the parser greedily
# consumes every subsequent token as part of the variadic list — so
# the prompt would be eaten as a tool name. Inserting `--` before
# the prompt terminates option parsing and leaves the prompt as a
# plain positional, which works for variadic and non-variadic flags
# alike.
cmd: List[str] = [
cli_path,
"--print",
@ -143,7 +154,9 @@ def run_claude(
]
if extra_args:
cmd.extend(extra_args)
cmd.append(prompt)
if prompt is not None:
cmd.append("--")
cmd.append(prompt)
# Build a minimal env for the CLI subprocess: only the allowlisted
# process-runtime vars from os.environ, plus the explicit proxy
@ -171,6 +184,7 @@ def run_claude(
completed = run_fn(
cmd,
env=env,
input=stdin_input,
capture_output=True,
text=True,
timeout=timeout,
@ -202,11 +216,12 @@ ModelResult = Union[DriverResult, ClaudeCLIError]
def run_claude_models_parallel(
*,
models: Sequence[str],
prompt: str,
prompt: Optional[str],
base_url: str,
api_key: str,
extra_env: Optional[Mapping[str, str]] = None,
extra_args: Optional[Sequence[str]] = None,
stdin_input: Optional[str] = None,
cli_path: str = CLAUDE_CLI_DEFAULT,
timeout: float = DEFAULT_TIMEOUT_SECONDS,
runner: Optional[Callable[..., Any]] = None,
@ -247,6 +262,7 @@ def run_claude_models_parallel(
api_key=api_key,
extra_env=extra_env,
extra_args=extra_args,
stdin_input=stdin_input,
cli_path=cli_path,
timeout=timeout,
runner=runner,

View file

@ -1,10 +1,10 @@
"""extended_thinking x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, and
assert that the upstream returned a `thinking` content block. This
proves the proxy preserves Anthropic's `thinking` request parameter
and the upstream response's `thinking` content blocks end-to-end.
to Anthropic, enable extended thinking via `--effort high`, and assert
that the upstream returned a `thinking` content block. This proves the
proxy preserves Anthropic's `thinking` request parameter and the
upstream response's `thinking` content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
@ -40,13 +40,23 @@ ANTHROPIC_MODELS = [
"claude-opus-4-7",
]
# A small budget is enough to surface a non-empty thinking block on
# even a trivial reasoning prompt; the test cares about wire shape, not
# answer quality.
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
# --effort max maps to the largest thinking budget on every supported
# Claude tier; the test cares about wire shape, not answer quality. We
# use a CLI flag (rather than the legacy MAX_THINKING_TOKENS env var)
# because Claude Code 2.x reads thinking config from --effort, not from
# the env, and silently no-ops the env var. We use `max` rather than
# `high` because Sonnet 4.6 / Opus 4.7 only emit thinking blocks when
# the budget is generous and the prompt is non-trivial.
THINKING_ARGS = ["--effort", "max"]
# A puzzle non-trivial enough that Sonnet/Opus actually engage thinking
# rather than answer from memory. Trivial arithmetic ("3-2=?") is
# optimized away on the modern tiers and arrives without a thinking
# block, which would make this test silently false-fail under
# `--effort max`. Haiku 4.5 thinks even for trivial prompts; Sonnet 4.6
# and Opus 4.7 only emit thinking when the upstream judges it useful.
THINKING_PROMPT = (
"Think step by step: if I have three apples and eat two, how many remain? "
"Answer with the single digit only."
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
@ -90,7 +100,7 @@ def test_extended_thinking_anthropic(compat_result):
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=THINKING_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models hosted in Microsoft Foundry on
Azure, enable extended thinking via `MAX_THINKING_TOKENS`, and assert
Azure, enable extended thinking via `--effort high`, and assert
that the upstream returned a `thinking` content block.
Foundry's Claude deployments advertise `supports_reasoning: true` in
@ -43,10 +43,10 @@ AZURE_MODELS = [
"claude-opus-4-7-azure",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"Think step by step: if I have three apples and eat two, how many remain? "
"Answer with the single digit only."
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
@ -88,7 +88,7 @@ def test_extended_thinking_azure(compat_result):
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=THINKING_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the unified `Converse` API path,
enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the
enable extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
@ -35,10 +35,10 @@ BEDROCK_CONVERSE_MODELS = [
"claude-opus-4-7-bedrock-converse",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"Think step by step: if I have three apples and eat two, how many remain? "
"Answer with the single digit only."
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_converse(compat_result):
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=THINKING_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the
enable extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
@ -35,10 +35,10 @@ BEDROCK_INVOKE_MODELS = [
"claude-opus-4-7-bedrock-invoke",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"Think step by step: if I have three apples and eat two, how many remain? "
"Answer with the single digit only."
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_invoke(compat_result):
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=THINKING_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models on Google Cloud Vertex AI, enable
extended thinking via `MAX_THINKING_TOKENS`, and assert that the
extended thinking via `--effort high`, and assert that the
upstream returned a `thinking` content block.
The (feature, provider) for this cell is inferred from the file path by
@ -35,10 +35,10 @@ VERTEX_AI_MODELS = [
"claude-opus-4-7-vertex",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_PROMPT = (
"Think step by step: if I have three apples and eat two, how many remain? "
"Answer with the single digit only."
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
"exactly 4 gallons of water? Think through the steps carefully."
)
@ -80,7 +80,7 @@ def test_extended_thinking_vertex_ai(compat_result):
prompt=THINKING_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=THINKING_ARGS,
)
failures = []

View file

@ -1,7 +1,7 @@
"""thinking_with_tool_use x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, allow
to Anthropic, enable extended thinking via `--effort high`, allow
the built-in `Bash` tool, and ask Claude to plan-and-execute a task
that requires both reasoning and a tool call. Assert the upstream
returned both a `thinking` content block and a `tool_use` content
@ -47,7 +47,7 @@ ANTHROPIC_MODELS = [
# Extended thinking on, with a small budget — enough to surface a
# non-empty thinking block on a trivial reasoning prompt without
# blowing up wall time.
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
# Prompt designed to force both blocks: the model has to *reason* about
# what command to run before *invoking* the Bash tool. Using a fixed
@ -105,8 +105,8 @@ def test_thinking_with_tool_use_anthropic(compat_result):
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=TOOL_USE_ARGS,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in
enable extended thinking via `--effort high`, allow the built-in
`Bash` tool, and ask Claude to plan-and-execute a task that requires
both reasoning and a tool call. Assert the upstream returned both a
`thinking` content block and a `tool_use` content block in the same
@ -38,7 +38,7 @@ AZURE_MODELS = [
"claude-opus-4-7-azure",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_TOOL_PROMPT = (
"Think step by step about which shell command would print just the word "
"'pong'. Then use the Bash tool to run that exact command and report what "
@ -86,8 +86,8 @@ def test_thinking_with_tool_use_azure(compat_result):
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=TOOL_USE_ARGS,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the `Converse` API, enable extended
thinking via `MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and
thinking via `--effort high`, allow the built-in `Bash` tool, and
ask Claude to plan-and-execute a task that requires both reasoning and
a tool call. Assert the upstream returned both a `thinking` content
block and a `tool_use` content block in the same turn.
@ -43,7 +43,7 @@ BEDROCK_CONVERSE_MODELS = [
"claude-opus-4-7-bedrock-converse",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_TOOL_PROMPT = (
"Think step by step about which shell command would print just the word "
"'pong'. Then use the Bash tool to run that exact command and report what "
@ -91,8 +91,8 @@ def test_thinking_with_tool_use_bedrock_converse(compat_result):
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=TOOL_USE_ARGS,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in
enable extended thinking via `--effort high`, allow the built-in
`Bash` tool, and ask Claude to plan-and-execute a task that requires
both reasoning and a tool call. Assert the upstream returned both a
`thinking` content block and a `tool_use` content block in the same
@ -45,7 +45,7 @@ BEDROCK_INVOKE_MODELS = [
"claude-opus-4-7-bedrock-invoke",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_TOOL_PROMPT = (
"Think step by step about which shell command would print just the word "
"'pong'. Then use the Bash tool to run that exact command and report what "
@ -93,8 +93,8 @@ def test_thinking_with_tool_use_bedrock_invoke(compat_result):
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=TOOL_USE_ARGS,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []

View file

@ -2,7 +2,7 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to GCP Vertex AI, enable extended thinking via
`MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and ask Claude
`--effort high`, allow the built-in `Bash` tool, and ask Claude
to plan-and-execute a task that requires both reasoning and a tool
call. Assert the upstream returned both a `thinking` content block and
a `tool_use` content block in the same turn.
@ -43,7 +43,7 @@ VERTEX_AI_MODELS = [
"claude-opus-4-7-vertex",
]
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
THINKING_ARGS = ["--effort", "max"]
THINKING_TOOL_PROMPT = (
"Think step by step about which shell command would print just the word "
"'pong'. Then use the Bash tool to run that exact command and report what "
@ -91,8 +91,8 @@ def test_thinking_with_tool_use_vertex_ai(compat_result):
prompt=THINKING_TOOL_PROMPT,
base_url=base_url,
api_key=api_key,
extra_env=THINKING_ENV,
extra_args=TOOL_USE_ARGS,
# thinking + tools combined into a single extra_args; see THINKING_ARGS
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
)
failures = []

View file

@ -1,10 +1,10 @@
"""vision x Anthropic.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, attach a small image via the CLI's `--image` flag, and
assert that the upstream produces a non-empty reply that references the
attached image. This proves the proxy preserves Claude Code's
multimodal content blocks end-to-end.
to Anthropic, attach a small image as an inline base64 `image` content
block via the CLI's `--input-format stream-json` mode, and assert that
the upstream produces a non-empty reply. This proves the proxy
preserves Claude Code's multimodal content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
@ -12,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by
tests/claude_code/vision/test_anthropic.py
^^^^^^ ^^^^^^^^^
feature_id provider
Why stream-json input rather than `--image <path>`: the claude CLI
dropped `--image` in 2.x. Image attachments are now driven via either
the Files API (server-uploaded blobs referenced by file_id) or by
sending an Anthropic-shaped user message through stdin. We use the
latter because it requires no upstream pre-upload — the test stays
hermetic and the wire shape (an `image` content block) is exactly what
the proxy must preserve.
"""
from __future__ import annotations
import base64
import json
import os
import pytest
@ -36,15 +44,50 @@ ANTHROPIC_MODELS = [
"claude-opus-4-7",
]
# Minimal 1x1 red PNG, base64-encoded. Decoded at test time and written
# to `tmp_path` so the CLI has a real file to attach without requiring
# any image-generation library or a checked-in binary fixture.
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
# `image` content block's source — no temp file or Files API upload
# needed, the test stays hermetic.
RED_PIXEL_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
VISION_PROMPT = (
"What single color do you see in the attached image? Answer in one word."
)
def test_vision_anthropic(compat_result, tmp_path):
def _build_stdin_input() -> str:
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
The CLI consumes a stream of `user` events whose `message.content` is
a list of Anthropic content blocks. A single user event with one
text block + one image block is enough to exercise the multimodal
code path.
"""
user_event = {
"type": "user",
"message": {
"role": "user",
"content": [
{"type": "text", "text": VISION_PROMPT},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": RED_PIXEL_PNG_B64,
},
},
],
},
}
return json.dumps(user_event) + "\n"
def test_vision_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with an image
attached and assert a non-empty reply."""
attached via stream-json input and assert a non-empty reply."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -61,15 +104,15 @@ def test_vision_anthropic(compat_result, tmp_path):
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
image_path = tmp_path / "red_pixel.png"
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
outcomes = run_claude_models_parallel(
models=ANTHROPIC_MODELS,
prompt="What single color do you see in the attached image? Answer in one word.",
# When using --input-format stream-json the CLI rejects a
# positional prompt; the prompt + image come in via stdin.
prompt=None,
base_url=base_url,
api_key=api_key,
extra_args=["--image", str(image_path)],
extra_args=["--input-format", "stream-json"],
stdin_input=_build_stdin_input(),
)
failures = []

View file

@ -1,26 +1,30 @@
"""vision x Azure (Microsoft Foundry).
"""vision x Azure.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models hosted in Microsoft Foundry on
Azure, attach a small image via the CLI's `--image` flag, and assert
that the upstream produces a non-empty reply.
Foundry's Anthropic deployments accept image content blocks identically
to anthropic.com (text + image input on Haiku 4.5, Sonnet 4.6, and Opus
4.7); LiteLLM passes them through unchanged on the `azure_ai/claude-*`
route.
to Azure, attach a small image as an inline base64 `image` content
block via the CLI's `--input-format stream-json` mode, and assert that
the upstream produces a non-empty reply. This proves the proxy
preserves Claude Code's multimodal content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/vision/test_azure.py
^^^^^^ ^^^^^
^^^^^^ ^^^^^^^^^
feature_id provider
Why stream-json input rather than `--image <path>`: the claude CLI
dropped `--image` in 2.x. Image attachments are now driven via either
the Files API (server-uploaded blobs referenced by file_id) or by
sending an Anthropic-shaped user message through stdin. We use the
latter because it requires no upstream pre-upload — the test stays
hermetic and the wire shape (an `image` content block) is exactly what
the proxy must preserve.
"""
from __future__ import annotations
import base64
import json
import os
import pytest
@ -40,12 +44,50 @@ AZURE_MODELS = [
"claude-opus-4-7-azure",
]
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
# `image` content block's source — no temp file or Files API upload
# needed, the test stays hermetic.
RED_PIXEL_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
VISION_PROMPT = (
"What single color do you see in the attached image? Answer in one word."
)
def test_vision_azure(compat_result, tmp_path):
def _build_stdin_input() -> str:
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
The CLI consumes a stream of `user` events whose `message.content` is
a list of Anthropic content blocks. A single user event with one
text block + one image block is enough to exercise the multimodal
code path.
"""
user_event = {
"type": "user",
"message": {
"role": "user",
"content": [
{"type": "text", "text": VISION_PROMPT},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": RED_PIXEL_PNG_B64,
},
},
],
},
}
return json.dumps(user_event) + "\n"
def test_vision_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with an image
attached and assert a non-empty reply."""
attached via stream-json input and assert a non-empty reply."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -62,15 +104,15 @@ def test_vision_azure(compat_result, tmp_path):
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
image_path = tmp_path / "red_pixel.png"
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
outcomes = run_claude_models_parallel(
models=AZURE_MODELS,
prompt="What single color do you see in the attached image? Answer in one word.",
# When using --input-format stream-json the CLI rejects a
# positional prompt; the prompt + image come in via stdin.
prompt=None,
base_url=base_url,
api_key=api_key,
extra_args=["--image", str(image_path)],
extra_args=["--input-format", "stream-json"],
stdin_input=_build_stdin_input(),
)
failures = []

View file

@ -1,21 +1,30 @@
"""vision x Bedrock (Converse).
"""vision x Bedrock Converse.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the unified `Converse` API path,
attach a small image via the CLI's `--image` flag, and assert that the
upstream produces a non-empty reply.
to Bedrock Converse, attach a small image as an inline base64 `image` content
block via the CLI's `--input-format stream-json` mode, and assert that
the upstream produces a non-empty reply. This proves the proxy
preserves Claude Code's multimodal content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/vision/test_bedrock_converse.py
^^^^^^ ^^^^^^^^^^^^^^^^
^^^^^^ ^^^^^^^^^
feature_id provider
Why stream-json input rather than `--image <path>`: the claude CLI
dropped `--image` in 2.x. Image attachments are now driven via either
the Files API (server-uploaded blobs referenced by file_id) or by
sending an Anthropic-shaped user message through stdin. We use the
latter because it requires no upstream pre-upload — the test stays
hermetic and the wire shape (an `image` content block) is exactly what
the proxy must preserve.
"""
from __future__ import annotations
import base64
import json
import os
import pytest
@ -35,12 +44,50 @@ BEDROCK_CONVERSE_MODELS = [
"claude-opus-4-7-bedrock-converse",
]
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
# `image` content block's source — no temp file or Files API upload
# needed, the test stays hermetic.
RED_PIXEL_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
VISION_PROMPT = (
"What single color do you see in the attached image? Answer in one word."
)
def test_vision_bedrock_converse(compat_result, tmp_path):
def _build_stdin_input() -> str:
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
The CLI consumes a stream of `user` events whose `message.content` is
a list of Anthropic content blocks. A single user event with one
text block + one image block is enough to exercise the multimodal
code path.
"""
user_event = {
"type": "user",
"message": {
"role": "user",
"content": [
{"type": "text", "text": VISION_PROMPT},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": RED_PIXEL_PNG_B64,
},
},
],
},
}
return json.dumps(user_event) + "\n"
def test_vision_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with an image
attached and assert a non-empty reply."""
attached via stream-json input and assert a non-empty reply."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -57,15 +104,15 @@ def test_vision_bedrock_converse(compat_result, tmp_path):
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
image_path = tmp_path / "red_pixel.png"
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
outcomes = run_claude_models_parallel(
models=BEDROCK_CONVERSE_MODELS,
prompt="What single color do you see in the attached image? Answer in one word.",
# When using --input-format stream-json the CLI rejects a
# positional prompt; the prompt + image come in via stdin.
prompt=None,
base_url=base_url,
api_key=api_key,
extra_args=["--image", str(image_path)],
extra_args=["--input-format", "stream-json"],
stdin_input=_build_stdin_input(),
)
failures = []

View file

@ -1,21 +1,30 @@
"""vision x Bedrock (Invoke).
"""vision x Bedrock Invoke.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
attach a small image via the CLI's `--image` flag, and assert that the
upstream produces a non-empty reply.
to Bedrock Invoke, attach a small image as an inline base64 `image` content
block via the CLI's `--input-format stream-json` mode, and assert that
the upstream produces a non-empty reply. This proves the proxy
preserves Claude Code's multimodal content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/vision/test_bedrock_invoke.py
^^^^^^ ^^^^^^^^^^^^^^
^^^^^^ ^^^^^^^^^
feature_id provider
Why stream-json input rather than `--image <path>`: the claude CLI
dropped `--image` in 2.x. Image attachments are now driven via either
the Files API (server-uploaded blobs referenced by file_id) or by
sending an Anthropic-shaped user message through stdin. We use the
latter because it requires no upstream pre-upload — the test stays
hermetic and the wire shape (an `image` content block) is exactly what
the proxy must preserve.
"""
from __future__ import annotations
import base64
import json
import os
import pytest
@ -35,12 +44,50 @@ BEDROCK_INVOKE_MODELS = [
"claude-opus-4-7-bedrock-invoke",
]
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
# `image` content block's source — no temp file or Files API upload
# needed, the test stays hermetic.
RED_PIXEL_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
VISION_PROMPT = (
"What single color do you see in the attached image? Answer in one word."
)
def test_vision_bedrock_invoke(compat_result, tmp_path):
def _build_stdin_input() -> str:
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
The CLI consumes a stream of `user` events whose `message.content` is
a list of Anthropic content blocks. A single user event with one
text block + one image block is enough to exercise the multimodal
code path.
"""
user_event = {
"type": "user",
"message": {
"role": "user",
"content": [
{"type": "text", "text": VISION_PROMPT},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": RED_PIXEL_PNG_B64,
},
},
],
},
}
return json.dumps(user_event) + "\n"
def test_vision_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with an image
attached and assert a non-empty reply."""
attached via stream-json input and assert a non-empty reply."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -57,15 +104,15 @@ def test_vision_bedrock_invoke(compat_result, tmp_path):
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
image_path = tmp_path / "red_pixel.png"
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
outcomes = run_claude_models_parallel(
models=BEDROCK_INVOKE_MODELS,
prompt="What single color do you see in the attached image? Answer in one word.",
# When using --input-format stream-json the CLI rejects a
# positional prompt; the prompt + image come in via stdin.
prompt=None,
base_url=base_url,
api_key=api_key,
extra_args=["--image", str(image_path)],
extra_args=["--input-format", "stream-json"],
stdin_input=_build_stdin_input(),
)
failures = []

View file

@ -1,9 +1,10 @@
"""vision x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Anthropic's models on Google Cloud Vertex AI, attach
a small image via the CLI's `--image` flag, and assert that the
upstream produces a non-empty reply.
to Vertex AI, attach a small image as an inline base64 `image` content
block via the CLI's `--input-format stream-json` mode, and assert that
the upstream produces a non-empty reply. This proves the proxy
preserves Claude Code's multimodal content blocks end-to-end.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
@ -11,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by
tests/claude_code/vision/test_vertex_ai.py
^^^^^^ ^^^^^^^^^
feature_id provider
Why stream-json input rather than `--image <path>`: the claude CLI
dropped `--image` in 2.x. Image attachments are now driven via either
the Files API (server-uploaded blobs referenced by file_id) or by
sending an Anthropic-shaped user message through stdin. We use the
latter because it requires no upstream pre-upload — the test stays
hermetic and the wire shape (an `image` content block) is exactly what
the proxy must preserve.
"""
from __future__ import annotations
import base64
import json
import os
import pytest
@ -35,12 +44,50 @@ VERTEX_AI_MODELS = [
"claude-opus-4-7-vertex",
]
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
# `image` content block's source — no temp file or Files API upload
# needed, the test stays hermetic.
RED_PIXEL_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
VISION_PROMPT = (
"What single color do you see in the attached image? Answer in one word."
)
def test_vision_vertex_ai(compat_result, tmp_path):
def _build_stdin_input() -> str:
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
The CLI consumes a stream of `user` events whose `message.content` is
a list of Anthropic content blocks. A single user event with one
text block + one image block is enough to exercise the multimodal
code path.
"""
user_event = {
"type": "user",
"message": {
"role": "user",
"content": [
{"type": "text", "text": VISION_PROMPT},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": RED_PIXEL_PNG_B64,
},
},
],
},
}
return json.dumps(user_event) + "\n"
def test_vision_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy with an image
attached and assert a non-empty reply."""
attached via stream-json input and assert a non-empty reply."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -57,15 +104,15 @@ def test_vision_vertex_ai(compat_result, tmp_path):
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
)
image_path = tmp_path / "red_pixel.png"
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
outcomes = run_claude_models_parallel(
models=VERTEX_AI_MODELS,
prompt="What single color do you see in the attached image? Answer in one word.",
# When using --input-format stream-json the CLI rejects a
# positional prompt; the prompt + image come in via stdin.
prompt=None,
base_url=base_url,
api_key=api_key,
extra_args=["--image", str(image_path)],
extra_args=["--input-format", "stream-json"],
stdin_input=_build_stdin_input(),
)
failures = []

View file

@ -3,19 +3,19 @@
Drive the real `claude` CLI against a running LiteLLM proxy that routes
to Anthropic, allow the built-in `WebSearch` tool, ask a question that
requires fresh web data, and assert that the upstream emitted a
`server_tool_use` or `web_search_tool_result` content block — proving
the proxy preserves Anthropic's server-side web search end-to-end.
`tool_use` block calling `WebSearch` — proving the proxy preserves
Claude Code's tool definitions and the upstream's tool-use response
end-to-end.
Web search is a *server tool*: unlike `Bash`/`Read`/etc., the upstream
executes the search itself and embeds the results inline in the
response. The wire shape is distinctive:
- `server_tool_use` block with `name: "web_search"`
- `web_search_tool_result` block carrying the encrypted result content
A regression where the proxy strips the `web_search_20250305` tool
from the request, drops the result block from the response, or fails
to forward the required beta header collapses both signals.
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
executes the search itself and feeds the result back as a `tool_result`
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
blocks (which only appear when the request includes the
`web_search_20250305` server tool definition — something the CLI does
not currently inject). A regression where the proxy strips the
`WebSearch` tool from the request or drops the `tool_use` block from
the response will break this assertion.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
@ -61,14 +61,13 @@ WEB_SEARCH_PROMPT = (
# answering from training data via a different tool.
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
# Block types that prove the server tool actually executed end-to-end.
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
WEB_SEARCH_TOOL_NAME = "WebSearch"
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a server-tool block (server_tool_use or
web_search_tool_result)."""
message included a `tool_use` block calling `WebSearch`."""
for event in events:
if event.get("type") != "assistant":
continue
@ -79,14 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
if (
block.get("type") == "tool_use"
and block.get("name") == WEB_SEARCH_TOOL_NAME
):
return True
return False
def test_web_search_anthropic(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream emitted a server-tool block proving web search ran."""
upstream emitted a `tool_use` block calling `WebSearch`, proving
the proxy preserved both the request-side tool definition and the
response-side tool_use block."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -126,11 +130,11 @@ def test_web_search_anthropic(compat_result):
failures.append(error)
continue
if not _has_server_tool_block(outcome.events):
if not _has_web_search_tool_use(outcome.events):
error = (
f"[{model}] no server_tool_use / web_search_tool_result block "
"observed; the proxy may have stripped the WebSearch server tool "
"or its result block"
f"[{model}] no `tool_use` block with name=WebSearch observed; "
"the proxy may have stripped the WebSearch tool definition from "
"the request or the tool_use block from the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)

View file

@ -1,16 +1,27 @@
"""web_search x Microsoft Foundry (Azure).
"""web_search x Azure.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
allow the built-in `WebSearch` tool, ask a question that requires
fresh web data, and assert that the upstream emitted a
`server_tool_use` or `web_search_tool_result` content block.
to Azure, allow the built-in `WebSearch` tool, ask a question that
requires fresh web data, and assert that the upstream emitted a
`tool_use` block calling `WebSearch` — proving the proxy preserves
Claude Code's tool definitions and the upstream's tool-use response
end-to-end.
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
executes the search itself and feeds the result back as a `tool_result`
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
blocks (which only appear when the request includes the
`web_search_20250305` server tool definition — something the CLI does
not currently inject). A regression where the proxy strips the
`WebSearch` tool from the request or drops the `tool_use` block from
the response will break this assertion.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/web_search/test_azure.py
^^^^^^^^^^ ^^^^^
^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
@ -36,15 +47,27 @@ AZURE_MODELS = [
"claude-opus-4-7-azure",
]
# A prompt the model cannot answer from training data alone — it forces
# the model to actually hit the web_search server tool rather than
# replying from memory. We pick "this week" as the freshness anchor
# because it's stable across long-running test schedules without
# pinning to a specific date that would go stale.
WEB_SEARCH_PROMPT = (
"Use web search to find a news headline published this week about "
"Anthropic. Reply with one sentence summarizing what you found."
)
# Allow only WebSearch so the model has no fallback path: if the proxy
# strips the server tool, the run will fail loudly rather than silently
# answering from training data via a different tool.
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
WEB_SEARCH_TOOL_NAME = "WebSearch"
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a `tool_use` block calling `WebSearch`."""
for event in events:
if event.get("type") != "assistant":
continue
@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
if (
block.get("type") == "tool_use"
and block.get("name") == WEB_SEARCH_TOOL_NAME
):
return True
return False
def test_web_search_azure(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream emitted a `tool_use` block calling `WebSearch`, proving
the proxy preserved both the request-side tool definition and the
response-side tool_use block."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -100,10 +130,11 @@ def test_web_search_azure(compat_result):
failures.append(error)
continue
if not _has_server_tool_block(outcome.events):
if not _has_web_search_tool_use(outcome.events):
error = (
f"[{model}] no server_tool_use / web_search_tool_result block "
"observed"
f"[{model}] no `tool_use` block with name=WebSearch observed; "
"the proxy may have stripped the WebSearch tool definition from "
"the request or the tool_use block from the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)

View file

@ -1,16 +1,27 @@
"""web_search x Bedrock (Converse).
"""web_search x Bedrock Converse.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the `Converse` API, allow the
built-in `WebSearch` tool, ask a question that requires fresh web
data, and assert that the upstream emitted a `server_tool_use` or
`web_search_tool_result` content block.
to Bedrock Converse, allow the built-in `WebSearch` tool, ask a question that
requires fresh web data, and assert that the upstream emitted a
`tool_use` block calling `WebSearch` — proving the proxy preserves
Claude Code's tool definitions and the upstream's tool-use response
end-to-end.
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
executes the search itself and feeds the result back as a `tool_result`
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
blocks (which only appear when the request includes the
`web_search_20250305` server tool definition — something the CLI does
not currently inject). A regression where the proxy strips the
`WebSearch` tool from the request or drops the `tool_use` block from
the response will break this assertion.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/web_search/test_bedrock_converse.py
^^^^^^^^^^ ^^^^^^^^^^^^^^^^
^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
@ -36,15 +47,27 @@ BEDROCK_CONVERSE_MODELS = [
"claude-opus-4-7-bedrock-converse",
]
# A prompt the model cannot answer from training data alone — it forces
# the model to actually hit the web_search server tool rather than
# replying from memory. We pick "this week" as the freshness anchor
# because it's stable across long-running test schedules without
# pinning to a specific date that would go stale.
WEB_SEARCH_PROMPT = (
"Use web search to find a news headline published this week about "
"Anthropic. Reply with one sentence summarizing what you found."
)
# Allow only WebSearch so the model has no fallback path: if the proxy
# strips the server tool, the run will fail loudly rather than silently
# answering from training data via a different tool.
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
WEB_SEARCH_TOOL_NAME = "WebSearch"
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a `tool_use` block calling `WebSearch`."""
for event in events:
if event.get("type") != "assistant":
continue
@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
if (
block.get("type") == "tool_use"
and block.get("name") == WEB_SEARCH_TOOL_NAME
):
return True
return False
def test_web_search_bedrock_converse(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream emitted a `tool_use` block calling `WebSearch`, proving
the proxy preserved both the request-side tool definition and the
response-side tool_use block."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -100,10 +130,11 @@ def test_web_search_bedrock_converse(compat_result):
failures.append(error)
continue
if not _has_server_tool_block(outcome.events):
if not _has_web_search_tool_use(outcome.events):
error = (
f"[{model}] no server_tool_use / web_search_tool_result block "
"observed"
f"[{model}] no `tool_use` block with name=WebSearch observed; "
"the proxy may have stripped the WebSearch tool definition from "
"the request or the tool_use block from the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)

View file

@ -1,22 +1,27 @@
"""web_search x Bedrock (Invoke).
"""web_search x Bedrock Invoke.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
allow the built-in `WebSearch` tool, ask a question that requires
fresh web data, and assert that the upstream emitted a
`server_tool_use` or `web_search_tool_result` content block.
to Bedrock Invoke, allow the built-in `WebSearch` tool, ask a question that
requires fresh web data, and assert that the upstream emitted a
`tool_use` block calling `WebSearch` — proving the proxy preserves
Claude Code's tool definitions and the upstream's tool-use response
end-to-end.
Bedrock support for the Anthropic-hosted web_search server tool has
been historically uneven — when it works, the wire shape is identical
to Anthropic's native API; when it doesn't, the upstream returns a
400 ("server tools not supported") that the proxy must surface
faithfully rather than silently dropping the tool from the request.
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
executes the search itself and feeds the result back as a `tool_result`
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
blocks (which only appear when the request includes the
`web_search_20250305` server tool definition — something the CLI does
not currently inject). A regression where the proxy strips the
`WebSearch` tool from the request or drops the `tool_use` block from
the response will break this assertion.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
tests/claude_code/web_search/test_bedrock_invoke.py
^^^^^^^^^^ ^^^^^^^^^^^^^^
^^^^^^^^^^ ^^^^^^^^^
feature_id provider
"""
@ -42,15 +47,27 @@ BEDROCK_INVOKE_MODELS = [
"claude-opus-4-7-bedrock-invoke",
]
# A prompt the model cannot answer from training data alone — it forces
# the model to actually hit the web_search server tool rather than
# replying from memory. We pick "this week" as the freshness anchor
# because it's stable across long-running test schedules without
# pinning to a specific date that would go stale.
WEB_SEARCH_PROMPT = (
"Use web search to find a news headline published this week about "
"Anthropic. Reply with one sentence summarizing what you found."
)
# Allow only WebSearch so the model has no fallback path: if the proxy
# strips the server tool, the run will fail loudly rather than silently
# answering from training data via a different tool.
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
WEB_SEARCH_TOOL_NAME = "WebSearch"
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a `tool_use` block calling `WebSearch`."""
for event in events:
if event.get("type") != "assistant":
continue
@ -61,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
if (
block.get("type") == "tool_use"
and block.get("name") == WEB_SEARCH_TOOL_NAME
):
return True
return False
def test_web_search_bedrock_invoke(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream emitted a `tool_use` block calling `WebSearch`, proving
the proxy preserved both the request-side tool definition and the
response-side tool_use block."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -106,11 +130,11 @@ def test_web_search_bedrock_invoke(compat_result):
failures.append(error)
continue
if not _has_server_tool_block(outcome.events):
if not _has_web_search_tool_use(outcome.events):
error = (
f"[{model}] no server_tool_use / web_search_tool_result block "
"observed; Bedrock may not support the web_search server tool "
"for this model, or the proxy stripped it"
f"[{model}] no `tool_use` block with name=WebSearch observed; "
"the proxy may have stripped the WebSearch tool definition from "
"the request or the tool_use block from the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)

View file

@ -1,15 +1,21 @@
"""web_search x Vertex AI.
Drive the real `claude` CLI against a running LiteLLM proxy that routes
Claude requests to GCP Vertex AI, allow the built-in `WebSearch` tool,
ask a question that requires fresh web data, and assert that the
upstream emitted a `server_tool_use` or `web_search_tool_result`
content block.
to Vertex AI, allow the built-in `WebSearch` tool, ask a question that
requires fresh web data, and assert that the upstream emitted a
`tool_use` block calling `WebSearch` — proving the proxy preserves
Claude Code's tool definitions and the upstream's tool-use response
end-to-end.
Per the changelog (1.0.110, 2.1.79+), Vertex has had inconsistent
support for the Anthropic web_search server tool — the cell will fail
loudly when the upstream rejects the tool, which is the diagnostic we
want for a compat matrix.
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
executes the search itself and feeds the result back as a `tool_result`
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
blocks (which only appear when the request includes the
`web_search_20250305` server tool definition — something the CLI does
not currently inject). A regression where the proxy strips the
`WebSearch` tool from the request or drops the `tool_use` block from
the response will break this assertion.
The (feature, provider) for this cell is inferred from the file path by
`tests/claude_code/conftest.py`:
@ -41,15 +47,27 @@ VERTEX_AI_MODELS = [
"claude-opus-4-7-vertex",
]
# A prompt the model cannot answer from training data alone — it forces
# the model to actually hit the web_search server tool rather than
# replying from memory. We pick "this week" as the freshness anchor
# because it's stable across long-running test schedules without
# pinning to a specific date that would go stale.
WEB_SEARCH_PROMPT = (
"Use web search to find a news headline published this week about "
"Anthropic. Reply with one sentence summarizing what you found."
)
# Allow only WebSearch so the model has no fallback path: if the proxy
# strips the server tool, the run will fail loudly rather than silently
# answering from training data via a different tool.
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
WEB_SEARCH_TOOL_NAME = "WebSearch"
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
"""Walk the stream-json events and return True if any assistant
message included a `tool_use` block calling `WebSearch`."""
for event in events:
if event.get("type") != "assistant":
continue
@ -60,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
for block in content:
if not isinstance(block, dict):
continue
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
if (
block.get("type") == "tool_use"
and block.get("name") == WEB_SEARCH_TOOL_NAME
):
return True
return False
def test_web_search_vertex_ai(compat_result):
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
upstream emitted a `tool_use` block calling `WebSearch`, proving
the proxy preserved both the request-side tool definition and the
response-side tool_use block."""
base_url = os.environ.get(PROXY_BASE_URL_ENV)
api_key = os.environ.get(PROXY_API_KEY_ENV)
if not base_url or not api_key:
@ -105,10 +130,11 @@ def test_web_search_vertex_ai(compat_result):
failures.append(error)
continue
if not _has_server_tool_block(outcome.events):
if not _has_web_search_tool_use(outcome.events):
error = (
f"[{model}] no server_tool_use / web_search_tool_result block "
"observed"
f"[{model}] no `tool_use` block with name=WebSearch observed; "
"the proxy may have stripped the WebSearch tool definition from "
"the request or the tool_use block from the response"
)
compat_result.add({"status": "fail", "error": error})
failures.append(error)