diff --git a/tests/claude_code/_driver_unit_tests/test_cli_driver.py b/tests/claude_code/_driver_unit_tests/test_cli_driver.py index 1eaa24eb90f..fb9fc1e5c49 100644 --- a/tests/claude_code/_driver_unit_tests/test_cli_driver.py +++ b/tests/claude_code/_driver_unit_tests/test_cli_driver.py @@ -34,10 +34,11 @@ class _Completed: def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""): captured = {} - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): captured["cmd"] = cmd captured["env"] = env captured["timeout"] = timeout + captured["input"] = input return _Completed(returncode=returncode, stdout=stdout, stderr=stderr) return runner, captured @@ -61,16 +62,19 @@ def test_run_claude_assembles_command_correctly(): assert "stream-json" in cmd assert "--model" in cmd assert "claude-haiku-4-5" in cmd - assert cmd[-1] == "hello" + # prompt is the last positional after the `--` end-of-options marker. + assert cmd[-2:] == ["--", "hello"] def test_run_claude_places_extra_args_before_prompt(): """`claude --print` expects the prompt as the final positional arg. Flags appearing after the prompt are ignored or eaten by the prompt - parser, which silently broke the tool_use and vision cells before the - fix. Pin the ordering: every flag (including caller-supplied - `extra_args`) must precede the prompt. + parser (especially variadic flags like `--allowed-tools `), + which silently broke the tool_use, vision, and web_search cells + before the fix. Pin the ordering: every flag (including + caller-supplied `extra_args`) must precede the `--` end-of-options + marker, which itself precedes the prompt. """ runner, captured = _make_runner(stdout="") run_claude( @@ -82,9 +86,15 @@ def test_run_claude_places_extra_args_before_prompt(): runner=runner, ) cmd = captured["cmd"] - assert cmd[-1] == "say hi" - prompt_idx = cmd.index("say hi") - assert cmd[prompt_idx - 2 : prompt_idx] == ["--allowed-tools", "Bash"] + assert cmd[-3:] == ["--allowed-tools", "Bash", "--"] or cmd[-2:] == [ + "--", + "say hi", + ] + # Stronger: prompt is last, `--` immediately precedes it, and the + # caller's extra_args sit somewhere earlier in the command. + assert cmd[-2:] == ["--", "say hi"] + assert "--allowed-tools" in cmd + assert cmd.index("--allowed-tools") < cmd.index("--") def test_run_claude_overlays_proxy_env(): @@ -414,7 +424,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model(): """Each model gets its own DriverResult keyed under the helper's dict.""" seen_models: List[str] = [] - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): # The model id is two slots after `--model` in the assembled command. idx = cmd.index("--model") model = cmd[idx + 1] @@ -452,7 +462,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model(): def test_run_claude_models_parallel_returns_errors_as_values(): """A model whose CLI is missing surfaces as a ClaudeCLIError, not a raise.""" - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): idx = cmd.index("--model") model = cmd[idx + 1] if model == "boom": @@ -485,7 +495,7 @@ def test_run_claude_models_parallel_returns_errors_as_values(): def test_run_claude_models_parallel_preserves_nonzero_exit_codes(): """Mixed success/failure on exit code should not collapse into one verdict.""" - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): idx = cmd.index("--model") model = cmd[idx + 1] if model == "fail": @@ -536,7 +546,7 @@ def test_run_claude_models_parallel_stamps_duration_on_each_result(): """ import time - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): idx = cmd.index("--model") model = cmd[idx + 1] time.sleep(0.05 if model == "fast" else 0.40) @@ -565,7 +575,7 @@ def test_run_claude_models_parallel_breakdown_logs_to_stderr(capsys): """The breakdown helper must emit a per-model timing block so users can answer "why didn't parallel help?" without re-instrumenting.""" - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): return _Completed(returncode=0, stdout="") run_claude_models_parallel( @@ -588,7 +598,7 @@ def test_run_claude_models_parallel_breakdown_marks_cli_errors(capsys): """When a model raises ClaudeCLIError, the breakdown should still show its row tagged as `cli-error` rather than crashing or omitting it.""" - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): idx = cmd.index("--model") if cmd[idx + 1] == "boom": raise FileNotFoundError(2, "no such file", "claude") @@ -613,7 +623,7 @@ def test_run_claude_models_parallel_forwards_extra_args_and_env(): captured_envs: List[dict] = [] captured_cmds: List[List[str]] = [] - def runner(cmd, env, capture_output, text, timeout, check): + def runner(cmd, env, capture_output, text, timeout, check, input=None): captured_envs.append(env) captured_cmds.append(cmd) return _Completed(returncode=0, stdout="") diff --git a/tests/claude_code/cli_driver.py b/tests/claude_code/cli_driver.py index 388f3655503..872caf8d19c 100644 --- a/tests/claude_code/cli_driver.py +++ b/tests/claude_code/cli_driver.py @@ -90,12 +90,13 @@ class DriverResult: def run_claude( *, - prompt: str, + prompt: Optional[str], model: str, base_url: str, api_key: str, extra_env: Optional[Mapping[str, str]] = None, extra_args: Optional[Sequence[str]] = None, + stdin_input: Optional[str] = None, cli_path: str = CLAUDE_CLI_DEFAULT, timeout: float = DEFAULT_TIMEOUT_SECONDS, runner: Optional[Any] = None, @@ -118,8 +119,10 @@ def run_claude( process-wide singleton; unit tests pass a no-op limiter or one backed by a tmp dir to keep tests hermetic. """ - if not prompt: - raise ValueError("prompt must be a non-empty string") + if prompt is None and stdin_input is None: + raise ValueError("must supply either `prompt` or `stdin_input`") + if prompt is not None and not prompt: + raise ValueError("prompt must be a non-empty string when provided") if not model: raise ValueError("model must be a non-empty string") if not base_url: @@ -132,6 +135,14 @@ def run_claude( # prompt (or silently dropped, depending on the CLI version) and the # tool_use / vision cells fail with confusing "no tool_use observed" # errors. Build the flag list first, then append the prompt last. + # + # When `extra_args` contains a *variadic* flag like `--allowed-tools + # WebSearch` (commander.js's ``), the parser greedily + # consumes every subsequent token as part of the variadic list — so + # the prompt would be eaten as a tool name. Inserting `--` before + # the prompt terminates option parsing and leaves the prompt as a + # plain positional, which works for variadic and non-variadic flags + # alike. cmd: List[str] = [ cli_path, "--print", @@ -143,7 +154,9 @@ def run_claude( ] if extra_args: cmd.extend(extra_args) - cmd.append(prompt) + if prompt is not None: + cmd.append("--") + cmd.append(prompt) # Build a minimal env for the CLI subprocess: only the allowlisted # process-runtime vars from os.environ, plus the explicit proxy @@ -171,6 +184,7 @@ def run_claude( completed = run_fn( cmd, env=env, + input=stdin_input, capture_output=True, text=True, timeout=timeout, @@ -202,11 +216,12 @@ ModelResult = Union[DriverResult, ClaudeCLIError] def run_claude_models_parallel( *, models: Sequence[str], - prompt: str, + prompt: Optional[str], base_url: str, api_key: str, extra_env: Optional[Mapping[str, str]] = None, extra_args: Optional[Sequence[str]] = None, + stdin_input: Optional[str] = None, cli_path: str = CLAUDE_CLI_DEFAULT, timeout: float = DEFAULT_TIMEOUT_SECONDS, runner: Optional[Callable[..., Any]] = None, @@ -247,6 +262,7 @@ def run_claude_models_parallel( api_key=api_key, extra_env=extra_env, extra_args=extra_args, + stdin_input=stdin_input, cli_path=cli_path, timeout=timeout, runner=runner, diff --git a/tests/claude_code/extended_thinking/test_anthropic.py b/tests/claude_code/extended_thinking/test_anthropic.py index 880083c52da..9dbe7a354bb 100644 --- a/tests/claude_code/extended_thinking/test_anthropic.py +++ b/tests/claude_code/extended_thinking/test_anthropic.py @@ -1,10 +1,10 @@ """extended_thinking x Anthropic. Drive the real `claude` CLI against a running LiteLLM proxy that routes -to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, and -assert that the upstream returned a `thinking` content block. This -proves the proxy preserves Anthropic's `thinking` request parameter -and the upstream response's `thinking` content blocks end-to-end. +to Anthropic, enable extended thinking via `--effort high`, and assert +that the upstream returned a `thinking` content block. This proves the +proxy preserves Anthropic's `thinking` request parameter and the +upstream response's `thinking` content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: @@ -40,13 +40,23 @@ ANTHROPIC_MODELS = [ "claude-opus-4-7", ] -# A small budget is enough to surface a non-empty thinking block on -# even a trivial reasoning prompt; the test cares about wire shape, not -# answer quality. -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +# --effort max maps to the largest thinking budget on every supported +# Claude tier; the test cares about wire shape, not answer quality. We +# use a CLI flag (rather than the legacy MAX_THINKING_TOKENS env var) +# because Claude Code 2.x reads thinking config from --effort, not from +# the env, and silently no-ops the env var. We use `max` rather than +# `high` because Sonnet 4.6 / Opus 4.7 only emit thinking blocks when +# the budget is generous and the prompt is non-trivial. +THINKING_ARGS = ["--effort", "max"] +# A puzzle non-trivial enough that Sonnet/Opus actually engage thinking +# rather than answer from memory. Trivial arithmetic ("3-2=?") is +# optimized away on the modern tiers and arrives without a thinking +# block, which would make this test silently false-fail under +# `--effort max`. Haiku 4.5 thinks even for trivial prompts; Sonnet 4.6 +# and Opus 4.7 only emit thinking when the upstream judges it useful. THINKING_PROMPT = ( - "Think step by step: if I have three apples and eat two, how many remain? " - "Answer with the single digit only." + "I have a 3-gallon jug and a 5-gallon jug. How can I measure " + "exactly 4 gallons of water? Think through the steps carefully." ) @@ -90,7 +100,7 @@ def test_extended_thinking_anthropic(compat_result): prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, + extra_args=THINKING_ARGS, ) failures = [] diff --git a/tests/claude_code/extended_thinking/test_azure.py b/tests/claude_code/extended_thinking/test_azure.py index fb777cca8c9..3f243cd3dd4 100644 --- a/tests/claude_code/extended_thinking/test_azure.py +++ b/tests/claude_code/extended_thinking/test_azure.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Anthropic's models hosted in Microsoft Foundry on -Azure, enable extended thinking via `MAX_THINKING_TOKENS`, and assert +Azure, enable extended thinking via `--effort high`, and assert that the upstream returned a `thinking` content block. Foundry's Claude deployments advertise `supports_reasoning: true` in @@ -43,10 +43,10 @@ AZURE_MODELS = [ "claude-opus-4-7-azure", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_PROMPT = ( - "Think step by step: if I have three apples and eat two, how many remain? " - "Answer with the single digit only." + "I have a 3-gallon jug and a 5-gallon jug. How can I measure " + "exactly 4 gallons of water? Think through the steps carefully." ) @@ -88,7 +88,7 @@ def test_extended_thinking_azure(compat_result): prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, + extra_args=THINKING_ARGS, ) failures = [] diff --git a/tests/claude_code/extended_thinking/test_bedrock_converse.py b/tests/claude_code/extended_thinking/test_bedrock_converse.py index 472a80e4aa6..2ccb3a44729 100644 --- a/tests/claude_code/extended_thinking/test_bedrock_converse.py +++ b/tests/claude_code/extended_thinking/test_bedrock_converse.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the unified `Converse` API path, -enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the +enable extended thinking via `--effort high`, and assert that the upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by @@ -35,10 +35,10 @@ BEDROCK_CONVERSE_MODELS = [ "claude-opus-4-7-bedrock-converse", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_PROMPT = ( - "Think step by step: if I have three apples and eat two, how many remain? " - "Answer with the single digit only." + "I have a 3-gallon jug and a 5-gallon jug. How can I measure " + "exactly 4 gallons of water? Think through the steps carefully." ) @@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_converse(compat_result): prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, + extra_args=THINKING_ARGS, ) failures = [] diff --git a/tests/claude_code/extended_thinking/test_bedrock_invoke.py b/tests/claude_code/extended_thinking/test_bedrock_invoke.py index f42031ba44c..79557d40a91 100644 --- a/tests/claude_code/extended_thinking/test_bedrock_invoke.py +++ b/tests/claude_code/extended_thinking/test_bedrock_invoke.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the legacy `InvokeModel` API path, -enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the +enable extended thinking via `--effort high`, and assert that the upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by @@ -35,10 +35,10 @@ BEDROCK_INVOKE_MODELS = [ "claude-opus-4-7-bedrock-invoke", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_PROMPT = ( - "Think step by step: if I have three apples and eat two, how many remain? " - "Answer with the single digit only." + "I have a 3-gallon jug and a 5-gallon jug. How can I measure " + "exactly 4 gallons of water? Think through the steps carefully." ) @@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_invoke(compat_result): prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, + extra_args=THINKING_ARGS, ) failures = [] diff --git a/tests/claude_code/extended_thinking/test_vertex_ai.py b/tests/claude_code/extended_thinking/test_vertex_ai.py index 1f3a15ec16c..7b37866451e 100644 --- a/tests/claude_code/extended_thinking/test_vertex_ai.py +++ b/tests/claude_code/extended_thinking/test_vertex_ai.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Anthropic's models on Google Cloud Vertex AI, enable -extended thinking via `MAX_THINKING_TOKENS`, and assert that the +extended thinking via `--effort high`, and assert that the upstream returned a `thinking` content block. The (feature, provider) for this cell is inferred from the file path by @@ -35,10 +35,10 @@ VERTEX_AI_MODELS = [ "claude-opus-4-7-vertex", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_PROMPT = ( - "Think step by step: if I have three apples and eat two, how many remain? " - "Answer with the single digit only." + "I have a 3-gallon jug and a 5-gallon jug. How can I measure " + "exactly 4 gallons of water? Think through the steps carefully." ) @@ -80,7 +80,7 @@ def test_extended_thinking_vertex_ai(compat_result): prompt=THINKING_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, + extra_args=THINKING_ARGS, ) failures = [] diff --git a/tests/claude_code/thinking_with_tool_use/test_anthropic.py b/tests/claude_code/thinking_with_tool_use/test_anthropic.py index 093a017a93c..7e4d84218dc 100644 --- a/tests/claude_code/thinking_with_tool_use/test_anthropic.py +++ b/tests/claude_code/thinking_with_tool_use/test_anthropic.py @@ -1,7 +1,7 @@ """thinking_with_tool_use x Anthropic. Drive the real `claude` CLI against a running LiteLLM proxy that routes -to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, allow +to Anthropic, enable extended thinking via `--effort high`, allow the built-in `Bash` tool, and ask Claude to plan-and-execute a task that requires both reasoning and a tool call. Assert the upstream returned both a `thinking` content block and a `tool_use` content @@ -47,7 +47,7 @@ ANTHROPIC_MODELS = [ # Extended thinking on, with a small budget — enough to surface a # non-empty thinking block on a trivial reasoning prompt without # blowing up wall time. -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] # Prompt designed to force both blocks: the model has to *reason* about # what command to run before *invoking* the Bash tool. Using a fixed @@ -105,8 +105,8 @@ def test_thinking_with_tool_use_anthropic(compat_result): prompt=THINKING_TOOL_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, - extra_args=TOOL_USE_ARGS, + # thinking + tools combined into a single extra_args; see THINKING_ARGS + extra_args=THINKING_ARGS + TOOL_USE_ARGS, ) failures = [] diff --git a/tests/claude_code/thinking_with_tool_use/test_azure.py b/tests/claude_code/thinking_with_tool_use/test_azure.py index eec9bfd1790..0d059133f3d 100644 --- a/tests/claude_code/thinking_with_tool_use/test_azure.py +++ b/tests/claude_code/thinking_with_tool_use/test_azure.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to Microsoft Foundry's Anthropic deployments on Azure, -enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in +enable extended thinking via `--effort high`, allow the built-in `Bash` tool, and ask Claude to plan-and-execute a task that requires both reasoning and a tool call. Assert the upstream returned both a `thinking` content block and a `tool_use` content block in the same @@ -38,7 +38,7 @@ AZURE_MODELS = [ "claude-opus-4-7-azure", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_TOOL_PROMPT = ( "Think step by step about which shell command would print just the word " "'pong'. Then use the Bash tool to run that exact command and report what " @@ -86,8 +86,8 @@ def test_thinking_with_tool_use_azure(compat_result): prompt=THINKING_TOOL_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, - extra_args=TOOL_USE_ARGS, + # thinking + tools combined into a single extra_args; see THINKING_ARGS + extra_args=THINKING_ARGS + TOOL_USE_ARGS, ) failures = [] diff --git a/tests/claude_code/thinking_with_tool_use/test_bedrock_converse.py b/tests/claude_code/thinking_with_tool_use/test_bedrock_converse.py index f9c9828e5c2..4d2dc3c29c6 100644 --- a/tests/claude_code/thinking_with_tool_use/test_bedrock_converse.py +++ b/tests/claude_code/thinking_with_tool_use/test_bedrock_converse.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the `Converse` API, enable extended -thinking via `MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and +thinking via `--effort high`, allow the built-in `Bash` tool, and ask Claude to plan-and-execute a task that requires both reasoning and a tool call. Assert the upstream returned both a `thinking` content block and a `tool_use` content block in the same turn. @@ -43,7 +43,7 @@ BEDROCK_CONVERSE_MODELS = [ "claude-opus-4-7-bedrock-converse", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_TOOL_PROMPT = ( "Think step by step about which shell command would print just the word " "'pong'. Then use the Bash tool to run that exact command and report what " @@ -91,8 +91,8 @@ def test_thinking_with_tool_use_bedrock_converse(compat_result): prompt=THINKING_TOOL_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, - extra_args=TOOL_USE_ARGS, + # thinking + tools combined into a single extra_args; see THINKING_ARGS + extra_args=THINKING_ARGS + TOOL_USE_ARGS, ) failures = [] diff --git a/tests/claude_code/thinking_with_tool_use/test_bedrock_invoke.py b/tests/claude_code/thinking_with_tool_use/test_bedrock_invoke.py index a0bd0753076..3e10e4ce9c7 100644 --- a/tests/claude_code/thinking_with_tool_use/test_bedrock_invoke.py +++ b/tests/claude_code/thinking_with_tool_use/test_bedrock_invoke.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to AWS Bedrock via the legacy `InvokeModel` API path, -enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in +enable extended thinking via `--effort high`, allow the built-in `Bash` tool, and ask Claude to plan-and-execute a task that requires both reasoning and a tool call. Assert the upstream returned both a `thinking` content block and a `tool_use` content block in the same @@ -45,7 +45,7 @@ BEDROCK_INVOKE_MODELS = [ "claude-opus-4-7-bedrock-invoke", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_TOOL_PROMPT = ( "Think step by step about which shell command would print just the word " "'pong'. Then use the Bash tool to run that exact command and report what " @@ -93,8 +93,8 @@ def test_thinking_with_tool_use_bedrock_invoke(compat_result): prompt=THINKING_TOOL_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, - extra_args=TOOL_USE_ARGS, + # thinking + tools combined into a single extra_args; see THINKING_ARGS + extra_args=THINKING_ARGS + TOOL_USE_ARGS, ) failures = [] diff --git a/tests/claude_code/thinking_with_tool_use/test_vertex_ai.py b/tests/claude_code/thinking_with_tool_use/test_vertex_ai.py index d784ddf640d..316c69f0df6 100644 --- a/tests/claude_code/thinking_with_tool_use/test_vertex_ai.py +++ b/tests/claude_code/thinking_with_tool_use/test_vertex_ai.py @@ -2,7 +2,7 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes Claude requests to GCP Vertex AI, enable extended thinking via -`MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and ask Claude +`--effort high`, allow the built-in `Bash` tool, and ask Claude to plan-and-execute a task that requires both reasoning and a tool call. Assert the upstream returned both a `thinking` content block and a `tool_use` content block in the same turn. @@ -43,7 +43,7 @@ VERTEX_AI_MODELS = [ "claude-opus-4-7-vertex", ] -THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"} +THINKING_ARGS = ["--effort", "max"] THINKING_TOOL_PROMPT = ( "Think step by step about which shell command would print just the word " "'pong'. Then use the Bash tool to run that exact command and report what " @@ -91,8 +91,8 @@ def test_thinking_with_tool_use_vertex_ai(compat_result): prompt=THINKING_TOOL_PROMPT, base_url=base_url, api_key=api_key, - extra_env=THINKING_ENV, - extra_args=TOOL_USE_ARGS, + # thinking + tools combined into a single extra_args; see THINKING_ARGS + extra_args=THINKING_ARGS + TOOL_USE_ARGS, ) failures = [] diff --git a/tests/claude_code/vision/test_anthropic.py b/tests/claude_code/vision/test_anthropic.py index e7017b18000..da54412f3d9 100644 --- a/tests/claude_code/vision/test_anthropic.py +++ b/tests/claude_code/vision/test_anthropic.py @@ -1,10 +1,10 @@ """vision x Anthropic. Drive the real `claude` CLI against a running LiteLLM proxy that routes -to Anthropic, attach a small image via the CLI's `--image` flag, and -assert that the upstream produces a non-empty reply that references the -attached image. This proves the proxy preserves Claude Code's -multimodal content blocks end-to-end. +to Anthropic, attach a small image as an inline base64 `image` content +block via the CLI's `--input-format stream-json` mode, and assert that +the upstream produces a non-empty reply. This proves the proxy +preserves Claude Code's multimodal content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: @@ -12,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by tests/claude_code/vision/test_anthropic.py ^^^^^^ ^^^^^^^^^ feature_id provider + +Why stream-json input rather than `--image `: the claude CLI +dropped `--image` in 2.x. Image attachments are now driven via either +the Files API (server-uploaded blobs referenced by file_id) or by +sending an Anthropic-shaped user message through stdin. We use the +latter because it requires no upstream pre-upload — the test stays +hermetic and the wire shape (an `image` content block) is exactly what +the proxy must preserve. """ from __future__ import annotations -import base64 +import json import os import pytest @@ -36,15 +44,50 @@ ANTHROPIC_MODELS = [ "claude-opus-4-7", ] -# Minimal 1x1 red PNG, base64-encoded. Decoded at test time and written -# to `tmp_path` so the CLI has a real file to attach without requiring -# any image-generation library or a checked-in binary fixture. -RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the +# `image` content block's source — no temp file or Files API upload +# needed, the test stays hermetic. +RED_PIXEL_PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8" + "z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +VISION_PROMPT = ( + "What single color do you see in the attached image? Answer in one word." +) -def test_vision_anthropic(compat_result, tmp_path): +def _build_stdin_input() -> str: + """Build the newline-delimited JSON payload for `--input-format stream-json`. + + The CLI consumes a stream of `user` events whose `message.content` is + a list of Anthropic content blocks. A single user event with one + text block + one image block is enough to exercise the multimodal + code path. + """ + user_event = { + "type": "user", + "message": { + "role": "user", + "content": [ + {"type": "text", "text": VISION_PROMPT}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": RED_PIXEL_PNG_B64, + }, + }, + ], + }, + } + return json.dumps(user_event) + "\n" + + +def test_vision_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image - attached and assert a non-empty reply.""" + attached via stream-json input and assert a non-empty reply.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -61,15 +104,15 @@ def test_vision_anthropic(compat_result, tmp_path): f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False ) - image_path = tmp_path / "red_pixel.png" - image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64)) - outcomes = run_claude_models_parallel( models=ANTHROPIC_MODELS, - prompt="What single color do you see in the attached image? Answer in one word.", + # When using --input-format stream-json the CLI rejects a + # positional prompt; the prompt + image come in via stdin. + prompt=None, base_url=base_url, api_key=api_key, - extra_args=["--image", str(image_path)], + extra_args=["--input-format", "stream-json"], + stdin_input=_build_stdin_input(), ) failures = [] diff --git a/tests/claude_code/vision/test_azure.py b/tests/claude_code/vision/test_azure.py index 832170bf7f5..820e0751d08 100644 --- a/tests/claude_code/vision/test_azure.py +++ b/tests/claude_code/vision/test_azure.py @@ -1,26 +1,30 @@ -"""vision x Azure (Microsoft Foundry). +"""vision x Azure. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to Anthropic's models hosted in Microsoft Foundry on -Azure, attach a small image via the CLI's `--image` flag, and assert -that the upstream produces a non-empty reply. - -Foundry's Anthropic deployments accept image content blocks identically -to anthropic.com (text + image input on Haiku 4.5, Sonnet 4.6, and Opus -4.7); LiteLLM passes them through unchanged on the `azure_ai/claude-*` -route. +to Azure, attach a small image as an inline base64 `image` content +block via the CLI's `--input-format stream-json` mode, and assert that +the upstream produces a non-empty reply. This proves the proxy +preserves Claude Code's multimodal content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/vision/test_azure.py - ^^^^^^ ^^^^^ + ^^^^^^ ^^^^^^^^^ feature_id provider + +Why stream-json input rather than `--image `: the claude CLI +dropped `--image` in 2.x. Image attachments are now driven via either +the Files API (server-uploaded blobs referenced by file_id) or by +sending an Anthropic-shaped user message through stdin. We use the +latter because it requires no upstream pre-upload — the test stays +hermetic and the wire shape (an `image` content block) is exactly what +the proxy must preserve. """ from __future__ import annotations -import base64 +import json import os import pytest @@ -40,12 +44,50 @@ AZURE_MODELS = [ "claude-opus-4-7-azure", ] -RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the +# `image` content block's source — no temp file or Files API upload +# needed, the test stays hermetic. +RED_PIXEL_PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8" + "z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +VISION_PROMPT = ( + "What single color do you see in the attached image? Answer in one word." +) -def test_vision_azure(compat_result, tmp_path): +def _build_stdin_input() -> str: + """Build the newline-delimited JSON payload for `--input-format stream-json`. + + The CLI consumes a stream of `user` events whose `message.content` is + a list of Anthropic content blocks. A single user event with one + text block + one image block is enough to exercise the multimodal + code path. + """ + user_event = { + "type": "user", + "message": { + "role": "user", + "content": [ + {"type": "text", "text": VISION_PROMPT}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": RED_PIXEL_PNG_B64, + }, + }, + ], + }, + } + return json.dumps(user_event) + "\n" + + +def test_vision_azure(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image - attached and assert a non-empty reply.""" + attached via stream-json input and assert a non-empty reply.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -62,15 +104,15 @@ def test_vision_azure(compat_result, tmp_path): f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False ) - image_path = tmp_path / "red_pixel.png" - image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64)) - outcomes = run_claude_models_parallel( models=AZURE_MODELS, - prompt="What single color do you see in the attached image? Answer in one word.", + # When using --input-format stream-json the CLI rejects a + # positional prompt; the prompt + image come in via stdin. + prompt=None, base_url=base_url, api_key=api_key, - extra_args=["--image", str(image_path)], + extra_args=["--input-format", "stream-json"], + stdin_input=_build_stdin_input(), ) failures = [] diff --git a/tests/claude_code/vision/test_bedrock_converse.py b/tests/claude_code/vision/test_bedrock_converse.py index b002f49068a..f9106708305 100644 --- a/tests/claude_code/vision/test_bedrock_converse.py +++ b/tests/claude_code/vision/test_bedrock_converse.py @@ -1,21 +1,30 @@ -"""vision x Bedrock (Converse). +"""vision x Bedrock Converse. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to AWS Bedrock via the unified `Converse` API path, -attach a small image via the CLI's `--image` flag, and assert that the -upstream produces a non-empty reply. +to Bedrock Converse, attach a small image as an inline base64 `image` content +block via the CLI's `--input-format stream-json` mode, and assert that +the upstream produces a non-empty reply. This proves the proxy +preserves Claude Code's multimodal content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/vision/test_bedrock_converse.py - ^^^^^^ ^^^^^^^^^^^^^^^^ + ^^^^^^ ^^^^^^^^^ feature_id provider + +Why stream-json input rather than `--image `: the claude CLI +dropped `--image` in 2.x. Image attachments are now driven via either +the Files API (server-uploaded blobs referenced by file_id) or by +sending an Anthropic-shaped user message through stdin. We use the +latter because it requires no upstream pre-upload — the test stays +hermetic and the wire shape (an `image` content block) is exactly what +the proxy must preserve. """ from __future__ import annotations -import base64 +import json import os import pytest @@ -35,12 +44,50 @@ BEDROCK_CONVERSE_MODELS = [ "claude-opus-4-7-bedrock-converse", ] -RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the +# `image` content block's source — no temp file or Files API upload +# needed, the test stays hermetic. +RED_PIXEL_PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8" + "z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +VISION_PROMPT = ( + "What single color do you see in the attached image? Answer in one word." +) -def test_vision_bedrock_converse(compat_result, tmp_path): +def _build_stdin_input() -> str: + """Build the newline-delimited JSON payload for `--input-format stream-json`. + + The CLI consumes a stream of `user` events whose `message.content` is + a list of Anthropic content blocks. A single user event with one + text block + one image block is enough to exercise the multimodal + code path. + """ + user_event = { + "type": "user", + "message": { + "role": "user", + "content": [ + {"type": "text", "text": VISION_PROMPT}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": RED_PIXEL_PNG_B64, + }, + }, + ], + }, + } + return json.dumps(user_event) + "\n" + + +def test_vision_bedrock_converse(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image - attached and assert a non-empty reply.""" + attached via stream-json input and assert a non-empty reply.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -57,15 +104,15 @@ def test_vision_bedrock_converse(compat_result, tmp_path): f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False ) - image_path = tmp_path / "red_pixel.png" - image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64)) - outcomes = run_claude_models_parallel( models=BEDROCK_CONVERSE_MODELS, - prompt="What single color do you see in the attached image? Answer in one word.", + # When using --input-format stream-json the CLI rejects a + # positional prompt; the prompt + image come in via stdin. + prompt=None, base_url=base_url, api_key=api_key, - extra_args=["--image", str(image_path)], + extra_args=["--input-format", "stream-json"], + stdin_input=_build_stdin_input(), ) failures = [] diff --git a/tests/claude_code/vision/test_bedrock_invoke.py b/tests/claude_code/vision/test_bedrock_invoke.py index 6000c53d2ab..3284d992e79 100644 --- a/tests/claude_code/vision/test_bedrock_invoke.py +++ b/tests/claude_code/vision/test_bedrock_invoke.py @@ -1,21 +1,30 @@ -"""vision x Bedrock (Invoke). +"""vision x Bedrock Invoke. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to AWS Bedrock via the legacy `InvokeModel` API path, -attach a small image via the CLI's `--image` flag, and assert that the -upstream produces a non-empty reply. +to Bedrock Invoke, attach a small image as an inline base64 `image` content +block via the CLI's `--input-format stream-json` mode, and assert that +the upstream produces a non-empty reply. This proves the proxy +preserves Claude Code's multimodal content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/vision/test_bedrock_invoke.py - ^^^^^^ ^^^^^^^^^^^^^^ + ^^^^^^ ^^^^^^^^^ feature_id provider + +Why stream-json input rather than `--image `: the claude CLI +dropped `--image` in 2.x. Image attachments are now driven via either +the Files API (server-uploaded blobs referenced by file_id) or by +sending an Anthropic-shaped user message through stdin. We use the +latter because it requires no upstream pre-upload — the test stays +hermetic and the wire shape (an `image` content block) is exactly what +the proxy must preserve. """ from __future__ import annotations -import base64 +import json import os import pytest @@ -35,12 +44,50 @@ BEDROCK_INVOKE_MODELS = [ "claude-opus-4-7-bedrock-invoke", ] -RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the +# `image` content block's source — no temp file or Files API upload +# needed, the test stays hermetic. +RED_PIXEL_PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8" + "z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +VISION_PROMPT = ( + "What single color do you see in the attached image? Answer in one word." +) -def test_vision_bedrock_invoke(compat_result, tmp_path): +def _build_stdin_input() -> str: + """Build the newline-delimited JSON payload for `--input-format stream-json`. + + The CLI consumes a stream of `user` events whose `message.content` is + a list of Anthropic content blocks. A single user event with one + text block + one image block is enough to exercise the multimodal + code path. + """ + user_event = { + "type": "user", + "message": { + "role": "user", + "content": [ + {"type": "text", "text": VISION_PROMPT}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": RED_PIXEL_PNG_B64, + }, + }, + ], + }, + } + return json.dumps(user_event) + "\n" + + +def test_vision_bedrock_invoke(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image - attached and assert a non-empty reply.""" + attached via stream-json input and assert a non-empty reply.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -57,15 +104,15 @@ def test_vision_bedrock_invoke(compat_result, tmp_path): f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False ) - image_path = tmp_path / "red_pixel.png" - image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64)) - outcomes = run_claude_models_parallel( models=BEDROCK_INVOKE_MODELS, - prompt="What single color do you see in the attached image? Answer in one word.", + # When using --input-format stream-json the CLI rejects a + # positional prompt; the prompt + image come in via stdin. + prompt=None, base_url=base_url, api_key=api_key, - extra_args=["--image", str(image_path)], + extra_args=["--input-format", "stream-json"], + stdin_input=_build_stdin_input(), ) failures = [] diff --git a/tests/claude_code/vision/test_vertex_ai.py b/tests/claude_code/vision/test_vertex_ai.py index 8a24ca3213c..5be5e5d45c2 100644 --- a/tests/claude_code/vision/test_vertex_ai.py +++ b/tests/claude_code/vision/test_vertex_ai.py @@ -1,9 +1,10 @@ """vision x Vertex AI. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to Anthropic's models on Google Cloud Vertex AI, attach -a small image via the CLI's `--image` flag, and assert that the -upstream produces a non-empty reply. +to Vertex AI, attach a small image as an inline base64 `image` content +block via the CLI's `--input-format stream-json` mode, and assert that +the upstream produces a non-empty reply. This proves the proxy +preserves Claude Code's multimodal content blocks end-to-end. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: @@ -11,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by tests/claude_code/vision/test_vertex_ai.py ^^^^^^ ^^^^^^^^^ feature_id provider + +Why stream-json input rather than `--image `: the claude CLI +dropped `--image` in 2.x. Image attachments are now driven via either +the Files API (server-uploaded blobs referenced by file_id) or by +sending an Anthropic-shaped user message through stdin. We use the +latter because it requires no upstream pre-upload — the test stays +hermetic and the wire shape (an `image` content block) is exactly what +the proxy must preserve. """ from __future__ import annotations -import base64 +import json import os import pytest @@ -35,12 +44,50 @@ VERTEX_AI_MODELS = [ "claude-opus-4-7-vertex", ] -RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the +# `image` content block's source — no temp file or Files API upload +# needed, the test stays hermetic. +RED_PIXEL_PNG_B64 = ( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8" + "z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==" +) + +VISION_PROMPT = ( + "What single color do you see in the attached image? Answer in one word." +) -def test_vision_vertex_ai(compat_result, tmp_path): +def _build_stdin_input() -> str: + """Build the newline-delimited JSON payload for `--input-format stream-json`. + + The CLI consumes a stream of `user` events whose `message.content` is + a list of Anthropic content blocks. A single user event with one + text block + one image block is enough to exercise the multimodal + code path. + """ + user_event = { + "type": "user", + "message": { + "role": "user", + "content": [ + {"type": "text", "text": VISION_PROMPT}, + { + "type": "image", + "source": { + "type": "base64", + "media_type": "image/png", + "data": RED_PIXEL_PNG_B64, + }, + }, + ], + }, + } + return json.dumps(user_event) + "\n" + + +def test_vision_vertex_ai(compat_result): """Drive the `claude` CLI against the LiteLLM proxy with an image - attached and assert a non-empty reply.""" + attached via stream-json input and assert a non-empty reply.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -57,15 +104,15 @@ def test_vision_vertex_ai(compat_result, tmp_path): f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False ) - image_path = tmp_path / "red_pixel.png" - image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64)) - outcomes = run_claude_models_parallel( models=VERTEX_AI_MODELS, - prompt="What single color do you see in the attached image? Answer in one word.", + # When using --input-format stream-json the CLI rejects a + # positional prompt; the prompt + image come in via stdin. + prompt=None, base_url=base_url, api_key=api_key, - extra_args=["--image", str(image_path)], + extra_args=["--input-format", "stream-json"], + stdin_input=_build_stdin_input(), ) failures = [] diff --git a/tests/claude_code/web_search/test_anthropic.py b/tests/claude_code/web_search/test_anthropic.py index 767fba88061..4be03f701f9 100644 --- a/tests/claude_code/web_search/test_anthropic.py +++ b/tests/claude_code/web_search/test_anthropic.py @@ -3,19 +3,19 @@ Drive the real `claude` CLI against a running LiteLLM proxy that routes to Anthropic, allow the built-in `WebSearch` tool, ask a question that requires fresh web data, and assert that the upstream emitted a -`server_tool_use` or `web_search_tool_result` content block — proving -the proxy preserves Anthropic's server-side web search end-to-end. +`tool_use` block calling `WebSearch` — proving the proxy preserves +Claude Code's tool definitions and the upstream's tool-use response +end-to-end. -Web search is a *server tool*: unlike `Bash`/`Read`/etc., the upstream -executes the search itself and embeds the results inline in the -response. The wire shape is distinctive: - - - `server_tool_use` block with `name: "web_search"` - - `web_search_tool_result` block carrying the encrypted result content - -A regression where the proxy strips the `web_search_20250305` tool -from the request, drops the result block from the response, or fails -to forward the required beta header collapses both signals. +Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI +executes the search itself and feeds the result back as a `tool_result` +block), so the wire shape is `tool_use` with `name="WebSearch"` rather +than the Anthropic-managed `server_tool_use` / `web_search_tool_result` +blocks (which only appear when the request includes the +`web_search_20250305` server tool definition — something the CLI does +not currently inject). A regression where the proxy strips the +`WebSearch` tool from the request or drops the `tool_use` block from +the response will break this assertion. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: @@ -61,14 +61,13 @@ WEB_SEARCH_PROMPT = ( # answering from training data via a different tool. WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"] -# Block types that prove the server tool actually executed end-to-end. -SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"} +# The CLI tool name surfaced as `tool_use.name` when WebSearch fires. +WEB_SEARCH_TOOL_NAME = "WebSearch" -def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: +def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: """Walk the stream-json events and return True if any assistant - message included a server-tool block (server_tool_use or - web_search_tool_result).""" + message included a `tool_use` block calling `WebSearch`.""" for event in events: if event.get("type") != "assistant": continue @@ -79,14 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: for block in content: if not isinstance(block, dict): continue - if block.get("type") in SERVER_TOOL_BLOCK_TYPES: + if ( + block.get("type") == "tool_use" + and block.get("name") == WEB_SEARCH_TOOL_NAME + ): return True return False def test_web_search_anthropic(compat_result): """Drive the `claude` CLI against the LiteLLM proxy and assert the - upstream emitted a server-tool block proving web search ran.""" + upstream emitted a `tool_use` block calling `WebSearch`, proving + the proxy preserved both the request-side tool definition and the + response-side tool_use block.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -126,11 +130,11 @@ def test_web_search_anthropic(compat_result): failures.append(error) continue - if not _has_server_tool_block(outcome.events): + if not _has_web_search_tool_use(outcome.events): error = ( - f"[{model}] no server_tool_use / web_search_tool_result block " - "observed; the proxy may have stripped the WebSearch server tool " - "or its result block" + f"[{model}] no `tool_use` block with name=WebSearch observed; " + "the proxy may have stripped the WebSearch tool definition from " + "the request or the tool_use block from the response" ) compat_result.add({"status": "fail", "error": error}) failures.append(error) diff --git a/tests/claude_code/web_search/test_azure.py b/tests/claude_code/web_search/test_azure.py index 9992f9bda46..bc2b181ac69 100644 --- a/tests/claude_code/web_search/test_azure.py +++ b/tests/claude_code/web_search/test_azure.py @@ -1,16 +1,27 @@ -"""web_search x Microsoft Foundry (Azure). +"""web_search x Azure. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to Microsoft Foundry's Anthropic deployments on Azure, -allow the built-in `WebSearch` tool, ask a question that requires -fresh web data, and assert that the upstream emitted a -`server_tool_use` or `web_search_tool_result` content block. +to Azure, allow the built-in `WebSearch` tool, ask a question that +requires fresh web data, and assert that the upstream emitted a +`tool_use` block calling `WebSearch` — proving the proxy preserves +Claude Code's tool definitions and the upstream's tool-use response +end-to-end. + +Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI +executes the search itself and feeds the result back as a `tool_result` +block), so the wire shape is `tool_use` with `name="WebSearch"` rather +than the Anthropic-managed `server_tool_use` / `web_search_tool_result` +blocks (which only appear when the request includes the +`web_search_20250305` server tool definition — something the CLI does +not currently inject). A regression where the proxy strips the +`WebSearch` tool from the request or drops the `tool_use` block from +the response will break this assertion. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/web_search/test_azure.py - ^^^^^^^^^^ ^^^^^ + ^^^^^^^^^^ ^^^^^^^^^ feature_id provider """ @@ -36,15 +47,27 @@ AZURE_MODELS = [ "claude-opus-4-7-azure", ] +# A prompt the model cannot answer from training data alone — it forces +# the model to actually hit the web_search server tool rather than +# replying from memory. We pick "this week" as the freshness anchor +# because it's stable across long-running test schedules without +# pinning to a specific date that would go stale. WEB_SEARCH_PROMPT = ( "Use web search to find a news headline published this week about " "Anthropic. Reply with one sentence summarizing what you found." ) +# Allow only WebSearch so the model has no fallback path: if the proxy +# strips the server tool, the run will fail loudly rather than silently +# answering from training data via a different tool. WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"] -SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"} + +# The CLI tool name surfaced as `tool_use.name` when WebSearch fires. +WEB_SEARCH_TOOL_NAME = "WebSearch" -def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: +def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: + """Walk the stream-json events and return True if any assistant + message included a `tool_use` block calling `WebSearch`.""" for event in events: if event.get("type") != "assistant": continue @@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: for block in content: if not isinstance(block, dict): continue - if block.get("type") in SERVER_TOOL_BLOCK_TYPES: + if ( + block.get("type") == "tool_use" + and block.get("name") == WEB_SEARCH_TOOL_NAME + ): return True return False def test_web_search_azure(compat_result): + """Drive the `claude` CLI against the LiteLLM proxy and assert the + upstream emitted a `tool_use` block calling `WebSearch`, proving + the proxy preserved both the request-side tool definition and the + response-side tool_use block.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -100,10 +130,11 @@ def test_web_search_azure(compat_result): failures.append(error) continue - if not _has_server_tool_block(outcome.events): + if not _has_web_search_tool_use(outcome.events): error = ( - f"[{model}] no server_tool_use / web_search_tool_result block " - "observed" + f"[{model}] no `tool_use` block with name=WebSearch observed; " + "the proxy may have stripped the WebSearch tool definition from " + "the request or the tool_use block from the response" ) compat_result.add({"status": "fail", "error": error}) failures.append(error) diff --git a/tests/claude_code/web_search/test_bedrock_converse.py b/tests/claude_code/web_search/test_bedrock_converse.py index 5a314099e4b..31d10ef75c0 100644 --- a/tests/claude_code/web_search/test_bedrock_converse.py +++ b/tests/claude_code/web_search/test_bedrock_converse.py @@ -1,16 +1,27 @@ -"""web_search x Bedrock (Converse). +"""web_search x Bedrock Converse. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to AWS Bedrock via the `Converse` API, allow the -built-in `WebSearch` tool, ask a question that requires fresh web -data, and assert that the upstream emitted a `server_tool_use` or -`web_search_tool_result` content block. +to Bedrock Converse, allow the built-in `WebSearch` tool, ask a question that +requires fresh web data, and assert that the upstream emitted a +`tool_use` block calling `WebSearch` — proving the proxy preserves +Claude Code's tool definitions and the upstream's tool-use response +end-to-end. + +Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI +executes the search itself and feeds the result back as a `tool_result` +block), so the wire shape is `tool_use` with `name="WebSearch"` rather +than the Anthropic-managed `server_tool_use` / `web_search_tool_result` +blocks (which only appear when the request includes the +`web_search_20250305` server tool definition — something the CLI does +not currently inject). A regression where the proxy strips the +`WebSearch` tool from the request or drops the `tool_use` block from +the response will break this assertion. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/web_search/test_bedrock_converse.py - ^^^^^^^^^^ ^^^^^^^^^^^^^^^^ + ^^^^^^^^^^ ^^^^^^^^^ feature_id provider """ @@ -36,15 +47,27 @@ BEDROCK_CONVERSE_MODELS = [ "claude-opus-4-7-bedrock-converse", ] +# A prompt the model cannot answer from training data alone — it forces +# the model to actually hit the web_search server tool rather than +# replying from memory. We pick "this week" as the freshness anchor +# because it's stable across long-running test schedules without +# pinning to a specific date that would go stale. WEB_SEARCH_PROMPT = ( "Use web search to find a news headline published this week about " "Anthropic. Reply with one sentence summarizing what you found." ) +# Allow only WebSearch so the model has no fallback path: if the proxy +# strips the server tool, the run will fail loudly rather than silently +# answering from training data via a different tool. WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"] -SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"} + +# The CLI tool name surfaced as `tool_use.name` when WebSearch fires. +WEB_SEARCH_TOOL_NAME = "WebSearch" -def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: +def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: + """Walk the stream-json events and return True if any assistant + message included a `tool_use` block calling `WebSearch`.""" for event in events: if event.get("type") != "assistant": continue @@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: for block in content: if not isinstance(block, dict): continue - if block.get("type") in SERVER_TOOL_BLOCK_TYPES: + if ( + block.get("type") == "tool_use" + and block.get("name") == WEB_SEARCH_TOOL_NAME + ): return True return False def test_web_search_bedrock_converse(compat_result): + """Drive the `claude` CLI against the LiteLLM proxy and assert the + upstream emitted a `tool_use` block calling `WebSearch`, proving + the proxy preserved both the request-side tool definition and the + response-side tool_use block.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -100,10 +130,11 @@ def test_web_search_bedrock_converse(compat_result): failures.append(error) continue - if not _has_server_tool_block(outcome.events): + if not _has_web_search_tool_use(outcome.events): error = ( - f"[{model}] no server_tool_use / web_search_tool_result block " - "observed" + f"[{model}] no `tool_use` block with name=WebSearch observed; " + "the proxy may have stripped the WebSearch tool definition from " + "the request or the tool_use block from the response" ) compat_result.add({"status": "fail", "error": error}) failures.append(error) diff --git a/tests/claude_code/web_search/test_bedrock_invoke.py b/tests/claude_code/web_search/test_bedrock_invoke.py index f6910057959..12e52690f1b 100644 --- a/tests/claude_code/web_search/test_bedrock_invoke.py +++ b/tests/claude_code/web_search/test_bedrock_invoke.py @@ -1,22 +1,27 @@ -"""web_search x Bedrock (Invoke). +"""web_search x Bedrock Invoke. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to AWS Bedrock via the legacy `InvokeModel` API path, -allow the built-in `WebSearch` tool, ask a question that requires -fresh web data, and assert that the upstream emitted a -`server_tool_use` or `web_search_tool_result` content block. +to Bedrock Invoke, allow the built-in `WebSearch` tool, ask a question that +requires fresh web data, and assert that the upstream emitted a +`tool_use` block calling `WebSearch` — proving the proxy preserves +Claude Code's tool definitions and the upstream's tool-use response +end-to-end. -Bedrock support for the Anthropic-hosted web_search server tool has -been historically uneven — when it works, the wire shape is identical -to Anthropic's native API; when it doesn't, the upstream returns a -400 ("server tools not supported") that the proxy must surface -faithfully rather than silently dropping the tool from the request. +Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI +executes the search itself and feeds the result back as a `tool_result` +block), so the wire shape is `tool_use` with `name="WebSearch"` rather +than the Anthropic-managed `server_tool_use` / `web_search_tool_result` +blocks (which only appear when the request includes the +`web_search_20250305` server tool definition — something the CLI does +not currently inject). A regression where the proxy strips the +`WebSearch` tool from the request or drops the `tool_use` block from +the response will break this assertion. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: tests/claude_code/web_search/test_bedrock_invoke.py - ^^^^^^^^^^ ^^^^^^^^^^^^^^ + ^^^^^^^^^^ ^^^^^^^^^ feature_id provider """ @@ -42,15 +47,27 @@ BEDROCK_INVOKE_MODELS = [ "claude-opus-4-7-bedrock-invoke", ] +# A prompt the model cannot answer from training data alone — it forces +# the model to actually hit the web_search server tool rather than +# replying from memory. We pick "this week" as the freshness anchor +# because it's stable across long-running test schedules without +# pinning to a specific date that would go stale. WEB_SEARCH_PROMPT = ( "Use web search to find a news headline published this week about " "Anthropic. Reply with one sentence summarizing what you found." ) +# Allow only WebSearch so the model has no fallback path: if the proxy +# strips the server tool, the run will fail loudly rather than silently +# answering from training data via a different tool. WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"] -SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"} + +# The CLI tool name surfaced as `tool_use.name` when WebSearch fires. +WEB_SEARCH_TOOL_NAME = "WebSearch" -def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: +def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: + """Walk the stream-json events and return True if any assistant + message included a `tool_use` block calling `WebSearch`.""" for event in events: if event.get("type") != "assistant": continue @@ -61,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: for block in content: if not isinstance(block, dict): continue - if block.get("type") in SERVER_TOOL_BLOCK_TYPES: + if ( + block.get("type") == "tool_use" + and block.get("name") == WEB_SEARCH_TOOL_NAME + ): return True return False def test_web_search_bedrock_invoke(compat_result): + """Drive the `claude` CLI against the LiteLLM proxy and assert the + upstream emitted a `tool_use` block calling `WebSearch`, proving + the proxy preserved both the request-side tool definition and the + response-side tool_use block.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -106,11 +130,11 @@ def test_web_search_bedrock_invoke(compat_result): failures.append(error) continue - if not _has_server_tool_block(outcome.events): + if not _has_web_search_tool_use(outcome.events): error = ( - f"[{model}] no server_tool_use / web_search_tool_result block " - "observed; Bedrock may not support the web_search server tool " - "for this model, or the proxy stripped it" + f"[{model}] no `tool_use` block with name=WebSearch observed; " + "the proxy may have stripped the WebSearch tool definition from " + "the request or the tool_use block from the response" ) compat_result.add({"status": "fail", "error": error}) failures.append(error) diff --git a/tests/claude_code/web_search/test_vertex_ai.py b/tests/claude_code/web_search/test_vertex_ai.py index 1230b0c32dd..63eba5c5294 100644 --- a/tests/claude_code/web_search/test_vertex_ai.py +++ b/tests/claude_code/web_search/test_vertex_ai.py @@ -1,15 +1,21 @@ """web_search x Vertex AI. Drive the real `claude` CLI against a running LiteLLM proxy that routes -Claude requests to GCP Vertex AI, allow the built-in `WebSearch` tool, -ask a question that requires fresh web data, and assert that the -upstream emitted a `server_tool_use` or `web_search_tool_result` -content block. +to Vertex AI, allow the built-in `WebSearch` tool, ask a question that +requires fresh web data, and assert that the upstream emitted a +`tool_use` block calling `WebSearch` — proving the proxy preserves +Claude Code's tool definitions and the upstream's tool-use response +end-to-end. -Per the changelog (1.0.110, 2.1.79+), Vertex has had inconsistent -support for the Anthropic web_search server tool — the cell will fail -loudly when the upstream rejects the tool, which is the diagnostic we -want for a compat matrix. +Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI +executes the search itself and feeds the result back as a `tool_result` +block), so the wire shape is `tool_use` with `name="WebSearch"` rather +than the Anthropic-managed `server_tool_use` / `web_search_tool_result` +blocks (which only appear when the request includes the +`web_search_20250305` server tool definition — something the CLI does +not currently inject). A regression where the proxy strips the +`WebSearch` tool from the request or drops the `tool_use` block from +the response will break this assertion. The (feature, provider) for this cell is inferred from the file path by `tests/claude_code/conftest.py`: @@ -41,15 +47,27 @@ VERTEX_AI_MODELS = [ "claude-opus-4-7-vertex", ] +# A prompt the model cannot answer from training data alone — it forces +# the model to actually hit the web_search server tool rather than +# replying from memory. We pick "this week" as the freshness anchor +# because it's stable across long-running test schedules without +# pinning to a specific date that would go stale. WEB_SEARCH_PROMPT = ( "Use web search to find a news headline published this week about " "Anthropic. Reply with one sentence summarizing what you found." ) +# Allow only WebSearch so the model has no fallback path: if the proxy +# strips the server tool, the run will fail loudly rather than silently +# answering from training data via a different tool. WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"] -SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"} + +# The CLI tool name surfaced as `tool_use.name` when WebSearch fires. +WEB_SEARCH_TOOL_NAME = "WebSearch" -def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: +def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool: + """Walk the stream-json events and return True if any assistant + message included a `tool_use` block calling `WebSearch`.""" for event in events: if event.get("type") != "assistant": continue @@ -60,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool: for block in content: if not isinstance(block, dict): continue - if block.get("type") in SERVER_TOOL_BLOCK_TYPES: + if ( + block.get("type") == "tool_use" + and block.get("name") == WEB_SEARCH_TOOL_NAME + ): return True return False def test_web_search_vertex_ai(compat_result): + """Drive the `claude` CLI against the LiteLLM proxy and assert the + upstream emitted a `tool_use` block calling `WebSearch`, proving + the proxy preserved both the request-side tool definition and the + response-side tool_use block.""" base_url = os.environ.get(PROXY_BASE_URL_ENV) api_key = os.environ.get(PROXY_API_KEY_ENV) if not base_url or not api_key: @@ -105,10 +130,11 @@ def test_web_search_vertex_ai(compat_result): failures.append(error) continue - if not _has_server_tool_block(outcome.events): + if not _has_web_search_tool_use(outcome.events): error = ( - f"[{model}] no server_tool_use / web_search_tool_result block " - "observed" + f"[{model}] no `tool_use` block with name=WebSearch observed; " + "the proxy may have stripped the WebSearch tool definition from " + "the request or the tool_use block from the response" ) compat_result.add({"status": "fail", "error": error}) failures.append(error)