mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
compat-matrix: fix vision, extended_thinking, web_search test bugs
These three cells were failing for reasons unrelated to LiteLLM translation: - vision: tests passed `--image <path>`, a flag that no longer exists in Claude Code 2.x (image attachment is now via the Files API or via `--input-format stream-json` with inline content blocks). Rewrite the cells to feed an Anthropic-shaped user message containing both text and a base64 `image` content block through stdin in stream-json mode. Hermetic — no temp file or Files API upload needed. - extended_thinking: tests set `MAX_THINKING_TOKENS=4096` as an env var, which Claude Code 2.x ignores. Switch to `--effort max` (the current CLI knob) and use a non-trivial prompt (3-gallon / 5-gallon jug puzzle). With trivial arithmetic the modern Sonnet/Opus tiers optimize away the thinking step and arrive without a thinking block, which made the test silently false-fail. - web_search: assertion looked for `server_tool_use` / `web_search_tool_result` blocks, but Claude Code's `WebSearch` is a *client-side* tool: the CLI executes the search itself and feeds the result back as a regular `tool_result` block. The Anthropic server-side `web_search_20250305` tool only fires when injected into the request directly (which the CLI does not do). Update the assertion to look for a `tool_use` block whose name is `WebSearch` — that's the right signal that the proxy preserved both the request-side tool definition and the response-side tool_use block end-to-end. Driver change required to support stream-json input + variadic flags: - cli_driver: insert `--` before the prompt positional. Variadic flags like `--allowed-tools <tools...>` (commander.js) greedily consume every following token, so the prompt was being eaten as a tool name and the CLI would error out with "Input must be provided either through stdin or as a prompt argument when using --print". - cli_driver: thread a `stdin_input` parameter through `run_claude` and `run_claude_models_parallel` so the vision rewrite can pipe stream-json events to the CLI on stdin. Validated end-to-end against a live LiteLLM proxy: all three Anthropic cells now pass on Haiku 4.5, Sonnet 4.6, and Opus 4.7. Driver unit tests (120) still green.
This commit is contained in:
parent
4caf8a6d37
commit
127bbd2b4c
22 changed files with 603 additions and 225 deletions
|
|
@ -34,10 +34,11 @@ class _Completed:
|
|||
def _make_runner(*, stdout: str = "", returncode: int = 0, stderr: str = ""):
|
||||
captured = {}
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
captured["cmd"] = cmd
|
||||
captured["env"] = env
|
||||
captured["timeout"] = timeout
|
||||
captured["input"] = input
|
||||
return _Completed(returncode=returncode, stdout=stdout, stderr=stderr)
|
||||
|
||||
return runner, captured
|
||||
|
|
@ -61,16 +62,19 @@ def test_run_claude_assembles_command_correctly():
|
|||
assert "stream-json" in cmd
|
||||
assert "--model" in cmd
|
||||
assert "claude-haiku-4-5" in cmd
|
||||
assert cmd[-1] == "hello"
|
||||
# prompt is the last positional after the `--` end-of-options marker.
|
||||
assert cmd[-2:] == ["--", "hello"]
|
||||
|
||||
|
||||
def test_run_claude_places_extra_args_before_prompt():
|
||||
"""`claude --print` expects the prompt as the final positional arg.
|
||||
|
||||
Flags appearing after the prompt are ignored or eaten by the prompt
|
||||
parser, which silently broke the tool_use and vision cells before the
|
||||
fix. Pin the ordering: every flag (including caller-supplied
|
||||
`extra_args`) must precede the prompt.
|
||||
parser (especially variadic flags like `--allowed-tools <tools...>`),
|
||||
which silently broke the tool_use, vision, and web_search cells
|
||||
before the fix. Pin the ordering: every flag (including
|
||||
caller-supplied `extra_args`) must precede the `--` end-of-options
|
||||
marker, which itself precedes the prompt.
|
||||
"""
|
||||
runner, captured = _make_runner(stdout="")
|
||||
run_claude(
|
||||
|
|
@ -82,9 +86,15 @@ def test_run_claude_places_extra_args_before_prompt():
|
|||
runner=runner,
|
||||
)
|
||||
cmd = captured["cmd"]
|
||||
assert cmd[-1] == "say hi"
|
||||
prompt_idx = cmd.index("say hi")
|
||||
assert cmd[prompt_idx - 2 : prompt_idx] == ["--allowed-tools", "Bash"]
|
||||
assert cmd[-3:] == ["--allowed-tools", "Bash", "--"] or cmd[-2:] == [
|
||||
"--",
|
||||
"say hi",
|
||||
]
|
||||
# Stronger: prompt is last, `--` immediately precedes it, and the
|
||||
# caller's extra_args sit somewhere earlier in the command.
|
||||
assert cmd[-2:] == ["--", "say hi"]
|
||||
assert "--allowed-tools" in cmd
|
||||
assert cmd.index("--allowed-tools") < cmd.index("--")
|
||||
|
||||
|
||||
def test_run_claude_overlays_proxy_env():
|
||||
|
|
@ -414,7 +424,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model():
|
|||
"""Each model gets its own DriverResult keyed under the helper's dict."""
|
||||
seen_models: List[str] = []
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
# The model id is two slots after `--model` in the assembled command.
|
||||
idx = cmd.index("--model")
|
||||
model = cmd[idx + 1]
|
||||
|
|
@ -452,7 +462,7 @@ def test_run_claude_models_parallel_returns_one_result_per_model():
|
|||
def test_run_claude_models_parallel_returns_errors_as_values():
|
||||
"""A model whose CLI is missing surfaces as a ClaudeCLIError, not a raise."""
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
idx = cmd.index("--model")
|
||||
model = cmd[idx + 1]
|
||||
if model == "boom":
|
||||
|
|
@ -485,7 +495,7 @@ def test_run_claude_models_parallel_returns_errors_as_values():
|
|||
def test_run_claude_models_parallel_preserves_nonzero_exit_codes():
|
||||
"""Mixed success/failure on exit code should not collapse into one verdict."""
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
idx = cmd.index("--model")
|
||||
model = cmd[idx + 1]
|
||||
if model == "fail":
|
||||
|
|
@ -536,7 +546,7 @@ def test_run_claude_models_parallel_stamps_duration_on_each_result():
|
|||
"""
|
||||
import time
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
idx = cmd.index("--model")
|
||||
model = cmd[idx + 1]
|
||||
time.sleep(0.05 if model == "fast" else 0.40)
|
||||
|
|
@ -565,7 +575,7 @@ def test_run_claude_models_parallel_breakdown_logs_to_stderr(capsys):
|
|||
"""The breakdown helper must emit a per-model timing block so users
|
||||
can answer "why didn't parallel help?" without re-instrumenting."""
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
return _Completed(returncode=0, stdout="")
|
||||
|
||||
run_claude_models_parallel(
|
||||
|
|
@ -588,7 +598,7 @@ def test_run_claude_models_parallel_breakdown_marks_cli_errors(capsys):
|
|||
"""When a model raises ClaudeCLIError, the breakdown should still
|
||||
show its row tagged as `cli-error` rather than crashing or omitting it."""
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
idx = cmd.index("--model")
|
||||
if cmd[idx + 1] == "boom":
|
||||
raise FileNotFoundError(2, "no such file", "claude")
|
||||
|
|
@ -613,7 +623,7 @@ def test_run_claude_models_parallel_forwards_extra_args_and_env():
|
|||
captured_envs: List[dict] = []
|
||||
captured_cmds: List[List[str]] = []
|
||||
|
||||
def runner(cmd, env, capture_output, text, timeout, check):
|
||||
def runner(cmd, env, capture_output, text, timeout, check, input=None):
|
||||
captured_envs.append(env)
|
||||
captured_cmds.append(cmd)
|
||||
return _Completed(returncode=0, stdout="")
|
||||
|
|
|
|||
|
|
@ -90,12 +90,13 @@ class DriverResult:
|
|||
|
||||
def run_claude(
|
||||
*,
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
model: str,
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
extra_env: Optional[Mapping[str, str]] = None,
|
||||
extra_args: Optional[Sequence[str]] = None,
|
||||
stdin_input: Optional[str] = None,
|
||||
cli_path: str = CLAUDE_CLI_DEFAULT,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
runner: Optional[Any] = None,
|
||||
|
|
@ -118,8 +119,10 @@ def run_claude(
|
|||
process-wide singleton; unit tests pass a no-op limiter or one
|
||||
backed by a tmp dir to keep tests hermetic.
|
||||
"""
|
||||
if not prompt:
|
||||
raise ValueError("prompt must be a non-empty string")
|
||||
if prompt is None and stdin_input is None:
|
||||
raise ValueError("must supply either `prompt` or `stdin_input`")
|
||||
if prompt is not None and not prompt:
|
||||
raise ValueError("prompt must be a non-empty string when provided")
|
||||
if not model:
|
||||
raise ValueError("model must be a non-empty string")
|
||||
if not base_url:
|
||||
|
|
@ -132,6 +135,14 @@ def run_claude(
|
|||
# prompt (or silently dropped, depending on the CLI version) and the
|
||||
# tool_use / vision cells fail with confusing "no tool_use observed"
|
||||
# errors. Build the flag list first, then append the prompt last.
|
||||
#
|
||||
# When `extra_args` contains a *variadic* flag like `--allowed-tools
|
||||
# WebSearch` (commander.js's `<tools...>`), the parser greedily
|
||||
# consumes every subsequent token as part of the variadic list — so
|
||||
# the prompt would be eaten as a tool name. Inserting `--` before
|
||||
# the prompt terminates option parsing and leaves the prompt as a
|
||||
# plain positional, which works for variadic and non-variadic flags
|
||||
# alike.
|
||||
cmd: List[str] = [
|
||||
cli_path,
|
||||
"--print",
|
||||
|
|
@ -143,7 +154,9 @@ def run_claude(
|
|||
]
|
||||
if extra_args:
|
||||
cmd.extend(extra_args)
|
||||
cmd.append(prompt)
|
||||
if prompt is not None:
|
||||
cmd.append("--")
|
||||
cmd.append(prompt)
|
||||
|
||||
# Build a minimal env for the CLI subprocess: only the allowlisted
|
||||
# process-runtime vars from os.environ, plus the explicit proxy
|
||||
|
|
@ -171,6 +184,7 @@ def run_claude(
|
|||
completed = run_fn(
|
||||
cmd,
|
||||
env=env,
|
||||
input=stdin_input,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=timeout,
|
||||
|
|
@ -202,11 +216,12 @@ ModelResult = Union[DriverResult, ClaudeCLIError]
|
|||
def run_claude_models_parallel(
|
||||
*,
|
||||
models: Sequence[str],
|
||||
prompt: str,
|
||||
prompt: Optional[str],
|
||||
base_url: str,
|
||||
api_key: str,
|
||||
extra_env: Optional[Mapping[str, str]] = None,
|
||||
extra_args: Optional[Sequence[str]] = None,
|
||||
stdin_input: Optional[str] = None,
|
||||
cli_path: str = CLAUDE_CLI_DEFAULT,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
runner: Optional[Callable[..., Any]] = None,
|
||||
|
|
@ -247,6 +262,7 @@ def run_claude_models_parallel(
|
|||
api_key=api_key,
|
||||
extra_env=extra_env,
|
||||
extra_args=extra_args,
|
||||
stdin_input=stdin_input,
|
||||
cli_path=cli_path,
|
||||
timeout=timeout,
|
||||
runner=runner,
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
"""extended_thinking x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, and
|
||||
assert that the upstream returned a `thinking` content block. This
|
||||
proves the proxy preserves Anthropic's `thinking` request parameter
|
||||
and the upstream response's `thinking` content blocks end-to-end.
|
||||
to Anthropic, enable extended thinking via `--effort high`, and assert
|
||||
that the upstream returned a `thinking` content block. This proves the
|
||||
proxy preserves Anthropic's `thinking` request parameter and the
|
||||
upstream response's `thinking` content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
|
@ -40,13 +40,23 @@ ANTHROPIC_MODELS = [
|
|||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
# A small budget is enough to surface a non-empty thinking block on
|
||||
# even a trivial reasoning prompt; the test cares about wire shape, not
|
||||
# answer quality.
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
# --effort max maps to the largest thinking budget on every supported
|
||||
# Claude tier; the test cares about wire shape, not answer quality. We
|
||||
# use a CLI flag (rather than the legacy MAX_THINKING_TOKENS env var)
|
||||
# because Claude Code 2.x reads thinking config from --effort, not from
|
||||
# the env, and silently no-ops the env var. We use `max` rather than
|
||||
# `high` because Sonnet 4.6 / Opus 4.7 only emit thinking blocks when
|
||||
# the budget is generous and the prompt is non-trivial.
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
# A puzzle non-trivial enough that Sonnet/Opus actually engage thinking
|
||||
# rather than answer from memory. Trivial arithmetic ("3-2=?") is
|
||||
# optimized away on the modern tiers and arrives without a thinking
|
||||
# block, which would make this test silently false-fail under
|
||||
# `--effort max`. Haiku 4.5 thinks even for trivial prompts; Sonnet 4.6
|
||||
# and Opus 4.7 only emit thinking when the upstream judges it useful.
|
||||
THINKING_PROMPT = (
|
||||
"Think step by step: if I have three apples and eat two, how many remain? "
|
||||
"Answer with the single digit only."
|
||||
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
|
||||
"exactly 4 gallons of water? Think through the steps carefully."
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -90,7 +100,7 @@ def test_extended_thinking_anthropic(compat_result):
|
|||
prompt=THINKING_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=THINKING_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models hosted in Microsoft Foundry on
|
||||
Azure, enable extended thinking via `MAX_THINKING_TOKENS`, and assert
|
||||
Azure, enable extended thinking via `--effort high`, and assert
|
||||
that the upstream returned a `thinking` content block.
|
||||
|
||||
Foundry's Claude deployments advertise `supports_reasoning: true` in
|
||||
|
|
@ -43,10 +43,10 @@ AZURE_MODELS = [
|
|||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_PROMPT = (
|
||||
"Think step by step: if I have three apples and eat two, how many remain? "
|
||||
"Answer with the single digit only."
|
||||
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
|
||||
"exactly 4 gallons of water? Think through the steps carefully."
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -88,7 +88,7 @@ def test_extended_thinking_azure(compat_result):
|
|||
prompt=THINKING_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=THINKING_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the unified `Converse` API path,
|
||||
enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the
|
||||
enable extended thinking via `--effort high`, and assert that the
|
||||
upstream returned a `thinking` content block.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
|
|
@ -35,10 +35,10 @@ BEDROCK_CONVERSE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_PROMPT = (
|
||||
"Think step by step: if I have three apples and eat two, how many remain? "
|
||||
"Answer with the single digit only."
|
||||
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
|
||||
"exactly 4 gallons of water? Think through the steps carefully."
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_converse(compat_result):
|
|||
prompt=THINKING_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=THINKING_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
||||
enable extended thinking via `MAX_THINKING_TOKENS`, and assert that the
|
||||
enable extended thinking via `--effort high`, and assert that the
|
||||
upstream returned a `thinking` content block.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
|
|
@ -35,10 +35,10 @@ BEDROCK_INVOKE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_PROMPT = (
|
||||
"Think step by step: if I have three apples and eat two, how many remain? "
|
||||
"Answer with the single digit only."
|
||||
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
|
||||
"exactly 4 gallons of water? Think through the steps carefully."
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -80,7 +80,7 @@ def test_extended_thinking_bedrock_invoke(compat_result):
|
|||
prompt=THINKING_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=THINKING_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models on Google Cloud Vertex AI, enable
|
||||
extended thinking via `MAX_THINKING_TOKENS`, and assert that the
|
||||
extended thinking via `--effort high`, and assert that the
|
||||
upstream returned a `thinking` content block.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
|
|
@ -35,10 +35,10 @@ VERTEX_AI_MODELS = [
|
|||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_PROMPT = (
|
||||
"Think step by step: if I have three apples and eat two, how many remain? "
|
||||
"Answer with the single digit only."
|
||||
"I have a 3-gallon jug and a 5-gallon jug. How can I measure "
|
||||
"exactly 4 gallons of water? Think through the steps carefully."
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -80,7 +80,7 @@ def test_extended_thinking_vertex_ai(compat_result):
|
|||
prompt=THINKING_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=THINKING_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""thinking_with_tool_use x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
to Anthropic, enable extended thinking via `MAX_THINKING_TOKENS`, allow
|
||||
to Anthropic, enable extended thinking via `--effort high`, allow
|
||||
the built-in `Bash` tool, and ask Claude to plan-and-execute a task
|
||||
that requires both reasoning and a tool call. Assert the upstream
|
||||
returned both a `thinking` content block and a `tool_use` content
|
||||
|
|
@ -47,7 +47,7 @@ ANTHROPIC_MODELS = [
|
|||
# Extended thinking on, with a small budget — enough to surface a
|
||||
# non-empty thinking block on a trivial reasoning prompt without
|
||||
# blowing up wall time.
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
|
||||
# Prompt designed to force both blocks: the model has to *reason* about
|
||||
# what command to run before *invoking* the Bash tool. Using a fixed
|
||||
|
|
@ -105,8 +105,8 @@ def test_thinking_with_tool_use_anthropic(compat_result):
|
|||
prompt=THINKING_TOOL_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=TOOL_USE_ARGS,
|
||||
# thinking + tools combined into a single extra_args; see THINKING_ARGS
|
||||
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
|
||||
enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in
|
||||
enable extended thinking via `--effort high`, allow the built-in
|
||||
`Bash` tool, and ask Claude to plan-and-execute a task that requires
|
||||
both reasoning and a tool call. Assert the upstream returned both a
|
||||
`thinking` content block and a `tool_use` content block in the same
|
||||
|
|
@ -38,7 +38,7 @@ AZURE_MODELS = [
|
|||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_TOOL_PROMPT = (
|
||||
"Think step by step about which shell command would print just the word "
|
||||
"'pong'. Then use the Bash tool to run that exact command and report what "
|
||||
|
|
@ -86,8 +86,8 @@ def test_thinking_with_tool_use_azure(compat_result):
|
|||
prompt=THINKING_TOOL_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=TOOL_USE_ARGS,
|
||||
# thinking + tools combined into a single extra_args; see THINKING_ARGS
|
||||
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the `Converse` API, enable extended
|
||||
thinking via `MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and
|
||||
thinking via `--effort high`, allow the built-in `Bash` tool, and
|
||||
ask Claude to plan-and-execute a task that requires both reasoning and
|
||||
a tool call. Assert the upstream returned both a `thinking` content
|
||||
block and a `tool_use` content block in the same turn.
|
||||
|
|
@ -43,7 +43,7 @@ BEDROCK_CONVERSE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_TOOL_PROMPT = (
|
||||
"Think step by step about which shell command would print just the word "
|
||||
"'pong'. Then use the Bash tool to run that exact command and report what "
|
||||
|
|
@ -91,8 +91,8 @@ def test_thinking_with_tool_use_bedrock_converse(compat_result):
|
|||
prompt=THINKING_TOOL_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=TOOL_USE_ARGS,
|
||||
# thinking + tools combined into a single extra_args; see THINKING_ARGS
|
||||
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
||||
enable extended thinking via `MAX_THINKING_TOKENS`, allow the built-in
|
||||
enable extended thinking via `--effort high`, allow the built-in
|
||||
`Bash` tool, and ask Claude to plan-and-execute a task that requires
|
||||
both reasoning and a tool call. Assert the upstream returned both a
|
||||
`thinking` content block and a `tool_use` content block in the same
|
||||
|
|
@ -45,7 +45,7 @@ BEDROCK_INVOKE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_TOOL_PROMPT = (
|
||||
"Think step by step about which shell command would print just the word "
|
||||
"'pong'. Then use the Bash tool to run that exact command and report what "
|
||||
|
|
@ -93,8 +93,8 @@ def test_thinking_with_tool_use_bedrock_invoke(compat_result):
|
|||
prompt=THINKING_TOOL_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=TOOL_USE_ARGS,
|
||||
# thinking + tools combined into a single extra_args; see THINKING_ARGS
|
||||
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to GCP Vertex AI, enable extended thinking via
|
||||
`MAX_THINKING_TOKENS`, allow the built-in `Bash` tool, and ask Claude
|
||||
`--effort high`, allow the built-in `Bash` tool, and ask Claude
|
||||
to plan-and-execute a task that requires both reasoning and a tool
|
||||
call. Assert the upstream returned both a `thinking` content block and
|
||||
a `tool_use` content block in the same turn.
|
||||
|
|
@ -43,7 +43,7 @@ VERTEX_AI_MODELS = [
|
|||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
THINKING_ENV = {"MAX_THINKING_TOKENS": "4096"}
|
||||
THINKING_ARGS = ["--effort", "max"]
|
||||
THINKING_TOOL_PROMPT = (
|
||||
"Think step by step about which shell command would print just the word "
|
||||
"'pong'. Then use the Bash tool to run that exact command and report what "
|
||||
|
|
@ -91,8 +91,8 @@ def test_thinking_with_tool_use_vertex_ai(compat_result):
|
|||
prompt=THINKING_TOOL_PROMPT,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_env=THINKING_ENV,
|
||||
extra_args=TOOL_USE_ARGS,
|
||||
# thinking + tools combined into a single extra_args; see THINKING_ARGS
|
||||
extra_args=THINKING_ARGS + TOOL_USE_ARGS,
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
"""vision x Anthropic.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
to Anthropic, attach a small image via the CLI's `--image` flag, and
|
||||
assert that the upstream produces a non-empty reply that references the
|
||||
attached image. This proves the proxy preserves Claude Code's
|
||||
multimodal content blocks end-to-end.
|
||||
to Anthropic, attach a small image as an inline base64 `image` content
|
||||
block via the CLI's `--input-format stream-json` mode, and assert that
|
||||
the upstream produces a non-empty reply. This proves the proxy
|
||||
preserves Claude Code's multimodal content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
|
@ -12,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by
|
|||
tests/claude_code/vision/test_anthropic.py
|
||||
^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why stream-json input rather than `--image <path>`: the claude CLI
|
||||
dropped `--image` in 2.x. Image attachments are now driven via either
|
||||
the Files API (server-uploaded blobs referenced by file_id) or by
|
||||
sending an Anthropic-shaped user message through stdin. We use the
|
||||
latter because it requires no upstream pre-upload — the test stays
|
||||
hermetic and the wire shape (an `image` content block) is exactly what
|
||||
the proxy must preserve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
|
@ -36,15 +44,50 @@ ANTHROPIC_MODELS = [
|
|||
"claude-opus-4-7",
|
||||
]
|
||||
|
||||
# Minimal 1x1 red PNG, base64-encoded. Decoded at test time and written
|
||||
# to `tmp_path` so the CLI has a real file to attach without requiring
|
||||
# any image-generation library or a checked-in binary fixture.
|
||||
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
|
||||
# `image` content block's source — no temp file or Files API upload
|
||||
# needed, the test stays hermetic.
|
||||
RED_PIXEL_PNG_B64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
|
||||
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
VISION_PROMPT = (
|
||||
"What single color do you see in the attached image? Answer in one word."
|
||||
)
|
||||
|
||||
|
||||
def test_vision_anthropic(compat_result, tmp_path):
|
||||
def _build_stdin_input() -> str:
|
||||
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
|
||||
|
||||
The CLI consumes a stream of `user` events whose `message.content` is
|
||||
a list of Anthropic content blocks. A single user event with one
|
||||
text block + one image block is enough to exercise the multimodal
|
||||
code path.
|
||||
"""
|
||||
user_event = {
|
||||
"type": "user",
|
||||
"message": {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": VISION_PROMPT},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": RED_PIXEL_PNG_B64,
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
return json.dumps(user_event) + "\n"
|
||||
|
||||
|
||||
def test_vision_anthropic(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with an image
|
||||
attached and assert a non-empty reply."""
|
||||
attached via stream-json input and assert a non-empty reply."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -61,15 +104,15 @@ def test_vision_anthropic(compat_result, tmp_path):
|
|||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
image_path = tmp_path / "red_pixel.png"
|
||||
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=ANTHROPIC_MODELS,
|
||||
prompt="What single color do you see in the attached image? Answer in one word.",
|
||||
# When using --input-format stream-json the CLI rejects a
|
||||
# positional prompt; the prompt + image come in via stdin.
|
||||
prompt=None,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--image", str(image_path)],
|
||||
extra_args=["--input-format", "stream-json"],
|
||||
stdin_input=_build_stdin_input(),
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,26 +1,30 @@
|
|||
"""vision x Azure (Microsoft Foundry).
|
||||
"""vision x Azure.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models hosted in Microsoft Foundry on
|
||||
Azure, attach a small image via the CLI's `--image` flag, and assert
|
||||
that the upstream produces a non-empty reply.
|
||||
|
||||
Foundry's Anthropic deployments accept image content blocks identically
|
||||
to anthropic.com (text + image input on Haiku 4.5, Sonnet 4.6, and Opus
|
||||
4.7); LiteLLM passes them through unchanged on the `azure_ai/claude-*`
|
||||
route.
|
||||
to Azure, attach a small image as an inline base64 `image` content
|
||||
block via the CLI's `--input-format stream-json` mode, and assert that
|
||||
the upstream produces a non-empty reply. This proves the proxy
|
||||
preserves Claude Code's multimodal content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/vision/test_azure.py
|
||||
^^^^^^ ^^^^^
|
||||
^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why stream-json input rather than `--image <path>`: the claude CLI
|
||||
dropped `--image` in 2.x. Image attachments are now driven via either
|
||||
the Files API (server-uploaded blobs referenced by file_id) or by
|
||||
sending an Anthropic-shaped user message through stdin. We use the
|
||||
latter because it requires no upstream pre-upload — the test stays
|
||||
hermetic and the wire shape (an `image` content block) is exactly what
|
||||
the proxy must preserve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
|
@ -40,12 +44,50 @@ AZURE_MODELS = [
|
|||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
|
||||
# `image` content block's source — no temp file or Files API upload
|
||||
# needed, the test stays hermetic.
|
||||
RED_PIXEL_PNG_B64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
|
||||
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
VISION_PROMPT = (
|
||||
"What single color do you see in the attached image? Answer in one word."
|
||||
)
|
||||
|
||||
|
||||
def test_vision_azure(compat_result, tmp_path):
|
||||
def _build_stdin_input() -> str:
|
||||
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
|
||||
|
||||
The CLI consumes a stream of `user` events whose `message.content` is
|
||||
a list of Anthropic content blocks. A single user event with one
|
||||
text block + one image block is enough to exercise the multimodal
|
||||
code path.
|
||||
"""
|
||||
user_event = {
|
||||
"type": "user",
|
||||
"message": {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": VISION_PROMPT},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": RED_PIXEL_PNG_B64,
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
return json.dumps(user_event) + "\n"
|
||||
|
||||
|
||||
def test_vision_azure(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with an image
|
||||
attached and assert a non-empty reply."""
|
||||
attached via stream-json input and assert a non-empty reply."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -62,15 +104,15 @@ def test_vision_azure(compat_result, tmp_path):
|
|||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
image_path = tmp_path / "red_pixel.png"
|
||||
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=AZURE_MODELS,
|
||||
prompt="What single color do you see in the attached image? Answer in one word.",
|
||||
# When using --input-format stream-json the CLI rejects a
|
||||
# positional prompt; the prompt + image come in via stdin.
|
||||
prompt=None,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--image", str(image_path)],
|
||||
extra_args=["--input-format", "stream-json"],
|
||||
stdin_input=_build_stdin_input(),
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,21 +1,30 @@
|
|||
"""vision x Bedrock (Converse).
|
||||
"""vision x Bedrock Converse.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the unified `Converse` API path,
|
||||
attach a small image via the CLI's `--image` flag, and assert that the
|
||||
upstream produces a non-empty reply.
|
||||
to Bedrock Converse, attach a small image as an inline base64 `image` content
|
||||
block via the CLI's `--input-format stream-json` mode, and assert that
|
||||
the upstream produces a non-empty reply. This proves the proxy
|
||||
preserves Claude Code's multimodal content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/vision/test_bedrock_converse.py
|
||||
^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why stream-json input rather than `--image <path>`: the claude CLI
|
||||
dropped `--image` in 2.x. Image attachments are now driven via either
|
||||
the Files API (server-uploaded blobs referenced by file_id) or by
|
||||
sending an Anthropic-shaped user message through stdin. We use the
|
||||
latter because it requires no upstream pre-upload — the test stays
|
||||
hermetic and the wire shape (an `image` content block) is exactly what
|
||||
the proxy must preserve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
|
@ -35,12 +44,50 @@ BEDROCK_CONVERSE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
|
||||
# `image` content block's source — no temp file or Files API upload
|
||||
# needed, the test stays hermetic.
|
||||
RED_PIXEL_PNG_B64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
|
||||
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
VISION_PROMPT = (
|
||||
"What single color do you see in the attached image? Answer in one word."
|
||||
)
|
||||
|
||||
|
||||
def test_vision_bedrock_converse(compat_result, tmp_path):
|
||||
def _build_stdin_input() -> str:
|
||||
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
|
||||
|
||||
The CLI consumes a stream of `user` events whose `message.content` is
|
||||
a list of Anthropic content blocks. A single user event with one
|
||||
text block + one image block is enough to exercise the multimodal
|
||||
code path.
|
||||
"""
|
||||
user_event = {
|
||||
"type": "user",
|
||||
"message": {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": VISION_PROMPT},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": RED_PIXEL_PNG_B64,
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
return json.dumps(user_event) + "\n"
|
||||
|
||||
|
||||
def test_vision_bedrock_converse(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with an image
|
||||
attached and assert a non-empty reply."""
|
||||
attached via stream-json input and assert a non-empty reply."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -57,15 +104,15 @@ def test_vision_bedrock_converse(compat_result, tmp_path):
|
|||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
image_path = tmp_path / "red_pixel.png"
|
||||
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_CONVERSE_MODELS,
|
||||
prompt="What single color do you see in the attached image? Answer in one word.",
|
||||
# When using --input-format stream-json the CLI rejects a
|
||||
# positional prompt; the prompt + image come in via stdin.
|
||||
prompt=None,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--image", str(image_path)],
|
||||
extra_args=["--input-format", "stream-json"],
|
||||
stdin_input=_build_stdin_input(),
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,21 +1,30 @@
|
|||
"""vision x Bedrock (Invoke).
|
||||
"""vision x Bedrock Invoke.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
||||
attach a small image via the CLI's `--image` flag, and assert that the
|
||||
upstream produces a non-empty reply.
|
||||
to Bedrock Invoke, attach a small image as an inline base64 `image` content
|
||||
block via the CLI's `--input-format stream-json` mode, and assert that
|
||||
the upstream produces a non-empty reply. This proves the proxy
|
||||
preserves Claude Code's multimodal content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/vision/test_bedrock_invoke.py
|
||||
^^^^^^ ^^^^^^^^^^^^^^
|
||||
^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why stream-json input rather than `--image <path>`: the claude CLI
|
||||
dropped `--image` in 2.x. Image attachments are now driven via either
|
||||
the Files API (server-uploaded blobs referenced by file_id) or by
|
||||
sending an Anthropic-shaped user message through stdin. We use the
|
||||
latter because it requires no upstream pre-upload — the test stays
|
||||
hermetic and the wire shape (an `image` content block) is exactly what
|
||||
the proxy must preserve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
|
@ -35,12 +44,50 @@ BEDROCK_INVOKE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
|
||||
# `image` content block's source — no temp file or Files API upload
|
||||
# needed, the test stays hermetic.
|
||||
RED_PIXEL_PNG_B64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
|
||||
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
VISION_PROMPT = (
|
||||
"What single color do you see in the attached image? Answer in one word."
|
||||
)
|
||||
|
||||
|
||||
def test_vision_bedrock_invoke(compat_result, tmp_path):
|
||||
def _build_stdin_input() -> str:
|
||||
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
|
||||
|
||||
The CLI consumes a stream of `user` events whose `message.content` is
|
||||
a list of Anthropic content blocks. A single user event with one
|
||||
text block + one image block is enough to exercise the multimodal
|
||||
code path.
|
||||
"""
|
||||
user_event = {
|
||||
"type": "user",
|
||||
"message": {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": VISION_PROMPT},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": RED_PIXEL_PNG_B64,
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
return json.dumps(user_event) + "\n"
|
||||
|
||||
|
||||
def test_vision_bedrock_invoke(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with an image
|
||||
attached and assert a non-empty reply."""
|
||||
attached via stream-json input and assert a non-empty reply."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -57,15 +104,15 @@ def test_vision_bedrock_invoke(compat_result, tmp_path):
|
|||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
image_path = tmp_path / "red_pixel.png"
|
||||
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=BEDROCK_INVOKE_MODELS,
|
||||
prompt="What single color do you see in the attached image? Answer in one word.",
|
||||
# When using --input-format stream-json the CLI rejects a
|
||||
# positional prompt; the prompt + image come in via stdin.
|
||||
prompt=None,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--image", str(image_path)],
|
||||
extra_args=["--input-format", "stream-json"],
|
||||
stdin_input=_build_stdin_input(),
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
"""vision x Vertex AI.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Anthropic's models on Google Cloud Vertex AI, attach
|
||||
a small image via the CLI's `--image` flag, and assert that the
|
||||
upstream produces a non-empty reply.
|
||||
to Vertex AI, attach a small image as an inline base64 `image` content
|
||||
block via the CLI's `--input-format stream-json` mode, and assert that
|
||||
the upstream produces a non-empty reply. This proves the proxy
|
||||
preserves Claude Code's multimodal content blocks end-to-end.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
|
@ -11,11 +12,19 @@ The (feature, provider) for this cell is inferred from the file path by
|
|||
tests/claude_code/vision/test_vertex_ai.py
|
||||
^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
|
||||
Why stream-json input rather than `--image <path>`: the claude CLI
|
||||
dropped `--image` in 2.x. Image attachments are now driven via either
|
||||
the Files API (server-uploaded blobs referenced by file_id) or by
|
||||
sending an Anthropic-shaped user message through stdin. We use the
|
||||
latter because it requires no upstream pre-upload — the test stays
|
||||
hermetic and the wire shape (an `image` content block) is exactly what
|
||||
the proxy must preserve.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
|
@ -35,12 +44,50 @@ VERTEX_AI_MODELS = [
|
|||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
RED_PIXEL_PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
# Minimal 1x1 red PNG, base64-encoded. We embed it directly as the
|
||||
# `image` content block's source — no temp file or Files API upload
|
||||
# needed, the test stays hermetic.
|
||||
RED_PIXEL_PNG_B64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8"
|
||||
"z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
||||
VISION_PROMPT = (
|
||||
"What single color do you see in the attached image? Answer in one word."
|
||||
)
|
||||
|
||||
|
||||
def test_vision_vertex_ai(compat_result, tmp_path):
|
||||
def _build_stdin_input() -> str:
|
||||
"""Build the newline-delimited JSON payload for `--input-format stream-json`.
|
||||
|
||||
The CLI consumes a stream of `user` events whose `message.content` is
|
||||
a list of Anthropic content blocks. A single user event with one
|
||||
text block + one image block is enough to exercise the multimodal
|
||||
code path.
|
||||
"""
|
||||
user_event = {
|
||||
"type": "user",
|
||||
"message": {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": VISION_PROMPT},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": RED_PIXEL_PNG_B64,
|
||||
},
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
return json.dumps(user_event) + "\n"
|
||||
|
||||
|
||||
def test_vision_vertex_ai(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy with an image
|
||||
attached and assert a non-empty reply."""
|
||||
attached via stream-json input and assert a non-empty reply."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -57,15 +104,15 @@ def test_vision_vertex_ai(compat_result, tmp_path):
|
|||
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
||||
)
|
||||
|
||||
image_path = tmp_path / "red_pixel.png"
|
||||
image_path.write_bytes(base64.b64decode(RED_PIXEL_PNG_B64))
|
||||
|
||||
outcomes = run_claude_models_parallel(
|
||||
models=VERTEX_AI_MODELS,
|
||||
prompt="What single color do you see in the attached image? Answer in one word.",
|
||||
# When using --input-format stream-json the CLI rejects a
|
||||
# positional prompt; the prompt + image come in via stdin.
|
||||
prompt=None,
|
||||
base_url=base_url,
|
||||
api_key=api_key,
|
||||
extra_args=["--image", str(image_path)],
|
||||
extra_args=["--input-format", "stream-json"],
|
||||
stdin_input=_build_stdin_input(),
|
||||
)
|
||||
|
||||
failures = []
|
||||
|
|
|
|||
|
|
@ -3,19 +3,19 @@
|
|||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
to Anthropic, allow the built-in `WebSearch` tool, ask a question that
|
||||
requires fresh web data, and assert that the upstream emitted a
|
||||
`server_tool_use` or `web_search_tool_result` content block — proving
|
||||
the proxy preserves Anthropic's server-side web search end-to-end.
|
||||
`tool_use` block calling `WebSearch` — proving the proxy preserves
|
||||
Claude Code's tool definitions and the upstream's tool-use response
|
||||
end-to-end.
|
||||
|
||||
Web search is a *server tool*: unlike `Bash`/`Read`/etc., the upstream
|
||||
executes the search itself and embeds the results inline in the
|
||||
response. The wire shape is distinctive:
|
||||
|
||||
- `server_tool_use` block with `name: "web_search"`
|
||||
- `web_search_tool_result` block carrying the encrypted result content
|
||||
|
||||
A regression where the proxy strips the `web_search_20250305` tool
|
||||
from the request, drops the result block from the response, or fails
|
||||
to forward the required beta header collapses both signals.
|
||||
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
|
||||
executes the search itself and feeds the result back as a `tool_result`
|
||||
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
|
||||
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
|
||||
blocks (which only appear when the request includes the
|
||||
`web_search_20250305` server tool definition — something the CLI does
|
||||
not currently inject). A regression where the proxy strips the
|
||||
`WebSearch` tool from the request or drops the `tool_use` block from
|
||||
the response will break this assertion.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
|
@ -61,14 +61,13 @@ WEB_SEARCH_PROMPT = (
|
|||
# answering from training data via a different tool.
|
||||
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
|
||||
|
||||
# Block types that prove the server tool actually executed end-to-end.
|
||||
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
|
||||
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
|
||||
WEB_SEARCH_TOOL_NAME = "WebSearch"
|
||||
|
||||
|
||||
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
"""Walk the stream-json events and return True if any assistant
|
||||
message included a server-tool block (server_tool_use or
|
||||
web_search_tool_result)."""
|
||||
message included a `tool_use` block calling `WebSearch`."""
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
|
|
@ -79,14 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
|
||||
if (
|
||||
block.get("type") == "tool_use"
|
||||
and block.get("name") == WEB_SEARCH_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_web_search_anthropic(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
|
||||
upstream emitted a server-tool block proving web search ran."""
|
||||
upstream emitted a `tool_use` block calling `WebSearch`, proving
|
||||
the proxy preserved both the request-side tool definition and the
|
||||
response-side tool_use block."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -126,11 +130,11 @@ def test_web_search_anthropic(compat_result):
|
|||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not _has_server_tool_block(outcome.events):
|
||||
if not _has_web_search_tool_use(outcome.events):
|
||||
error = (
|
||||
f"[{model}] no server_tool_use / web_search_tool_result block "
|
||||
"observed; the proxy may have stripped the WebSearch server tool "
|
||||
"or its result block"
|
||||
f"[{model}] no `tool_use` block with name=WebSearch observed; "
|
||||
"the proxy may have stripped the WebSearch tool definition from "
|
||||
"the request or the tool_use block from the response"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
|
|
|
|||
|
|
@ -1,16 +1,27 @@
|
|||
"""web_search x Microsoft Foundry (Azure).
|
||||
"""web_search x Azure.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to Microsoft Foundry's Anthropic deployments on Azure,
|
||||
allow the built-in `WebSearch` tool, ask a question that requires
|
||||
fresh web data, and assert that the upstream emitted a
|
||||
`server_tool_use` or `web_search_tool_result` content block.
|
||||
to Azure, allow the built-in `WebSearch` tool, ask a question that
|
||||
requires fresh web data, and assert that the upstream emitted a
|
||||
`tool_use` block calling `WebSearch` — proving the proxy preserves
|
||||
Claude Code's tool definitions and the upstream's tool-use response
|
||||
end-to-end.
|
||||
|
||||
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
|
||||
executes the search itself and feeds the result back as a `tool_result`
|
||||
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
|
||||
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
|
||||
blocks (which only appear when the request includes the
|
||||
`web_search_20250305` server tool definition — something the CLI does
|
||||
not currently inject). A regression where the proxy strips the
|
||||
`WebSearch` tool from the request or drops the `tool_use` block from
|
||||
the response will break this assertion.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/web_search/test_azure.py
|
||||
^^^^^^^^^^ ^^^^^
|
||||
^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
|
|
@ -36,15 +47,27 @@ AZURE_MODELS = [
|
|||
"claude-opus-4-7-azure",
|
||||
]
|
||||
|
||||
# A prompt the model cannot answer from training data alone — it forces
|
||||
# the model to actually hit the web_search server tool rather than
|
||||
# replying from memory. We pick "this week" as the freshness anchor
|
||||
# because it's stable across long-running test schedules without
|
||||
# pinning to a specific date that would go stale.
|
||||
WEB_SEARCH_PROMPT = (
|
||||
"Use web search to find a news headline published this week about "
|
||||
"Anthropic. Reply with one sentence summarizing what you found."
|
||||
)
|
||||
# Allow only WebSearch so the model has no fallback path: if the proxy
|
||||
# strips the server tool, the run will fail loudly rather than silently
|
||||
# answering from training data via a different tool.
|
||||
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
|
||||
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
|
||||
|
||||
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
|
||||
WEB_SEARCH_TOOL_NAME = "WebSearch"
|
||||
|
||||
|
||||
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
"""Walk the stream-json events and return True if any assistant
|
||||
message included a `tool_use` block calling `WebSearch`."""
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
|
|
@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
|
||||
if (
|
||||
block.get("type") == "tool_use"
|
||||
and block.get("name") == WEB_SEARCH_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_web_search_azure(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
|
||||
upstream emitted a `tool_use` block calling `WebSearch`, proving
|
||||
the proxy preserved both the request-side tool definition and the
|
||||
response-side tool_use block."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -100,10 +130,11 @@ def test_web_search_azure(compat_result):
|
|||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not _has_server_tool_block(outcome.events):
|
||||
if not _has_web_search_tool_use(outcome.events):
|
||||
error = (
|
||||
f"[{model}] no server_tool_use / web_search_tool_result block "
|
||||
"observed"
|
||||
f"[{model}] no `tool_use` block with name=WebSearch observed; "
|
||||
"the proxy may have stripped the WebSearch tool definition from "
|
||||
"the request or the tool_use block from the response"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
|
|
|
|||
|
|
@ -1,16 +1,27 @@
|
|||
"""web_search x Bedrock (Converse).
|
||||
"""web_search x Bedrock Converse.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the `Converse` API, allow the
|
||||
built-in `WebSearch` tool, ask a question that requires fresh web
|
||||
data, and assert that the upstream emitted a `server_tool_use` or
|
||||
`web_search_tool_result` content block.
|
||||
to Bedrock Converse, allow the built-in `WebSearch` tool, ask a question that
|
||||
requires fresh web data, and assert that the upstream emitted a
|
||||
`tool_use` block calling `WebSearch` — proving the proxy preserves
|
||||
Claude Code's tool definitions and the upstream's tool-use response
|
||||
end-to-end.
|
||||
|
||||
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
|
||||
executes the search itself and feeds the result back as a `tool_result`
|
||||
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
|
||||
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
|
||||
blocks (which only appear when the request includes the
|
||||
`web_search_20250305` server tool definition — something the CLI does
|
||||
not currently inject). A regression where the proxy strips the
|
||||
`WebSearch` tool from the request or drops the `tool_use` block from
|
||||
the response will break this assertion.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/web_search/test_bedrock_converse.py
|
||||
^^^^^^^^^^ ^^^^^^^^^^^^^^^^
|
||||
^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
|
|
@ -36,15 +47,27 @@ BEDROCK_CONVERSE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-converse",
|
||||
]
|
||||
|
||||
# A prompt the model cannot answer from training data alone — it forces
|
||||
# the model to actually hit the web_search server tool rather than
|
||||
# replying from memory. We pick "this week" as the freshness anchor
|
||||
# because it's stable across long-running test schedules without
|
||||
# pinning to a specific date that would go stale.
|
||||
WEB_SEARCH_PROMPT = (
|
||||
"Use web search to find a news headline published this week about "
|
||||
"Anthropic. Reply with one sentence summarizing what you found."
|
||||
)
|
||||
# Allow only WebSearch so the model has no fallback path: if the proxy
|
||||
# strips the server tool, the run will fail loudly rather than silently
|
||||
# answering from training data via a different tool.
|
||||
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
|
||||
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
|
||||
|
||||
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
|
||||
WEB_SEARCH_TOOL_NAME = "WebSearch"
|
||||
|
||||
|
||||
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
"""Walk the stream-json events and return True if any assistant
|
||||
message included a `tool_use` block calling `WebSearch`."""
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
|
|
@ -55,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
|
||||
if (
|
||||
block.get("type") == "tool_use"
|
||||
and block.get("name") == WEB_SEARCH_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_web_search_bedrock_converse(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
|
||||
upstream emitted a `tool_use` block calling `WebSearch`, proving
|
||||
the proxy preserved both the request-side tool definition and the
|
||||
response-side tool_use block."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -100,10 +130,11 @@ def test_web_search_bedrock_converse(compat_result):
|
|||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not _has_server_tool_block(outcome.events):
|
||||
if not _has_web_search_tool_use(outcome.events):
|
||||
error = (
|
||||
f"[{model}] no server_tool_use / web_search_tool_result block "
|
||||
"observed"
|
||||
f"[{model}] no `tool_use` block with name=WebSearch observed; "
|
||||
"the proxy may have stripped the WebSearch tool definition from "
|
||||
"the request or the tool_use block from the response"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
|
|
|
|||
|
|
@ -1,22 +1,27 @@
|
|||
"""web_search x Bedrock (Invoke).
|
||||
"""web_search x Bedrock Invoke.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
||||
allow the built-in `WebSearch` tool, ask a question that requires
|
||||
fresh web data, and assert that the upstream emitted a
|
||||
`server_tool_use` or `web_search_tool_result` content block.
|
||||
to Bedrock Invoke, allow the built-in `WebSearch` tool, ask a question that
|
||||
requires fresh web data, and assert that the upstream emitted a
|
||||
`tool_use` block calling `WebSearch` — proving the proxy preserves
|
||||
Claude Code's tool definitions and the upstream's tool-use response
|
||||
end-to-end.
|
||||
|
||||
Bedrock support for the Anthropic-hosted web_search server tool has
|
||||
been historically uneven — when it works, the wire shape is identical
|
||||
to Anthropic's native API; when it doesn't, the upstream returns a
|
||||
400 ("server tools not supported") that the proxy must surface
|
||||
faithfully rather than silently dropping the tool from the request.
|
||||
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
|
||||
executes the search itself and feeds the result back as a `tool_result`
|
||||
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
|
||||
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
|
||||
blocks (which only appear when the request includes the
|
||||
`web_search_20250305` server tool definition — something the CLI does
|
||||
not currently inject). A regression where the proxy strips the
|
||||
`WebSearch` tool from the request or drops the `tool_use` block from
|
||||
the response will break this assertion.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
||||
tests/claude_code/web_search/test_bedrock_invoke.py
|
||||
^^^^^^^^^^ ^^^^^^^^^^^^^^
|
||||
^^^^^^^^^^ ^^^^^^^^^
|
||||
feature_id provider
|
||||
"""
|
||||
|
||||
|
|
@ -42,15 +47,27 @@ BEDROCK_INVOKE_MODELS = [
|
|||
"claude-opus-4-7-bedrock-invoke",
|
||||
]
|
||||
|
||||
# A prompt the model cannot answer from training data alone — it forces
|
||||
# the model to actually hit the web_search server tool rather than
|
||||
# replying from memory. We pick "this week" as the freshness anchor
|
||||
# because it's stable across long-running test schedules without
|
||||
# pinning to a specific date that would go stale.
|
||||
WEB_SEARCH_PROMPT = (
|
||||
"Use web search to find a news headline published this week about "
|
||||
"Anthropic. Reply with one sentence summarizing what you found."
|
||||
)
|
||||
# Allow only WebSearch so the model has no fallback path: if the proxy
|
||||
# strips the server tool, the run will fail loudly rather than silently
|
||||
# answering from training data via a different tool.
|
||||
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
|
||||
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
|
||||
|
||||
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
|
||||
WEB_SEARCH_TOOL_NAME = "WebSearch"
|
||||
|
||||
|
||||
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
"""Walk the stream-json events and return True if any assistant
|
||||
message included a `tool_use` block calling `WebSearch`."""
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
|
|
@ -61,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
|
||||
if (
|
||||
block.get("type") == "tool_use"
|
||||
and block.get("name") == WEB_SEARCH_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_web_search_bedrock_invoke(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
|
||||
upstream emitted a `tool_use` block calling `WebSearch`, proving
|
||||
the proxy preserved both the request-side tool definition and the
|
||||
response-side tool_use block."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -106,11 +130,11 @@ def test_web_search_bedrock_invoke(compat_result):
|
|||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not _has_server_tool_block(outcome.events):
|
||||
if not _has_web_search_tool_use(outcome.events):
|
||||
error = (
|
||||
f"[{model}] no server_tool_use / web_search_tool_result block "
|
||||
"observed; Bedrock may not support the web_search server tool "
|
||||
"for this model, or the proxy stripped it"
|
||||
f"[{model}] no `tool_use` block with name=WebSearch observed; "
|
||||
"the proxy may have stripped the WebSearch tool definition from "
|
||||
"the request or the tool_use block from the response"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
|
|
|
|||
|
|
@ -1,15 +1,21 @@
|
|||
"""web_search x Vertex AI.
|
||||
|
||||
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
||||
Claude requests to GCP Vertex AI, allow the built-in `WebSearch` tool,
|
||||
ask a question that requires fresh web data, and assert that the
|
||||
upstream emitted a `server_tool_use` or `web_search_tool_result`
|
||||
content block.
|
||||
to Vertex AI, allow the built-in `WebSearch` tool, ask a question that
|
||||
requires fresh web data, and assert that the upstream emitted a
|
||||
`tool_use` block calling `WebSearch` — proving the proxy preserves
|
||||
Claude Code's tool definitions and the upstream's tool-use response
|
||||
end-to-end.
|
||||
|
||||
Per the changelog (1.0.110, 2.1.79+), Vertex has had inconsistent
|
||||
support for the Anthropic web_search server tool — the cell will fail
|
||||
loudly when the upstream rejects the tool, which is the diagnostic we
|
||||
want for a compat matrix.
|
||||
Note: Claude Code's `WebSearch` is a *client-side* tool (the CLI
|
||||
executes the search itself and feeds the result back as a `tool_result`
|
||||
block), so the wire shape is `tool_use` with `name="WebSearch"` rather
|
||||
than the Anthropic-managed `server_tool_use` / `web_search_tool_result`
|
||||
blocks (which only appear when the request includes the
|
||||
`web_search_20250305` server tool definition — something the CLI does
|
||||
not currently inject). A regression where the proxy strips the
|
||||
`WebSearch` tool from the request or drops the `tool_use` block from
|
||||
the response will break this assertion.
|
||||
|
||||
The (feature, provider) for this cell is inferred from the file path by
|
||||
`tests/claude_code/conftest.py`:
|
||||
|
|
@ -41,15 +47,27 @@ VERTEX_AI_MODELS = [
|
|||
"claude-opus-4-7-vertex",
|
||||
]
|
||||
|
||||
# A prompt the model cannot answer from training data alone — it forces
|
||||
# the model to actually hit the web_search server tool rather than
|
||||
# replying from memory. We pick "this week" as the freshness anchor
|
||||
# because it's stable across long-running test schedules without
|
||||
# pinning to a specific date that would go stale.
|
||||
WEB_SEARCH_PROMPT = (
|
||||
"Use web search to find a news headline published this week about "
|
||||
"Anthropic. Reply with one sentence summarizing what you found."
|
||||
)
|
||||
# Allow only WebSearch so the model has no fallback path: if the proxy
|
||||
# strips the server tool, the run will fail loudly rather than silently
|
||||
# answering from training data via a different tool.
|
||||
WEB_SEARCH_ARGS = ["--allowed-tools", "WebSearch"]
|
||||
SERVER_TOOL_BLOCK_TYPES = {"server_tool_use", "web_search_tool_result"}
|
||||
|
||||
# The CLI tool name surfaced as `tool_use.name` when WebSearch fires.
|
||||
WEB_SEARCH_TOOL_NAME = "WebSearch"
|
||||
|
||||
|
||||
def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
def _has_web_search_tool_use(events: Sequence[Mapping[str, Any]]) -> bool:
|
||||
"""Walk the stream-json events and return True if any assistant
|
||||
message included a `tool_use` block calling `WebSearch`."""
|
||||
for event in events:
|
||||
if event.get("type") != "assistant":
|
||||
continue
|
||||
|
|
@ -60,12 +78,19 @@ def _has_server_tool_block(events: Sequence[Mapping[str, Any]]) -> bool:
|
|||
for block in content:
|
||||
if not isinstance(block, dict):
|
||||
continue
|
||||
if block.get("type") in SERVER_TOOL_BLOCK_TYPES:
|
||||
if (
|
||||
block.get("type") == "tool_use"
|
||||
and block.get("name") == WEB_SEARCH_TOOL_NAME
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_web_search_vertex_ai(compat_result):
|
||||
"""Drive the `claude` CLI against the LiteLLM proxy and assert the
|
||||
upstream emitted a `tool_use` block calling `WebSearch`, proving
|
||||
the proxy preserved both the request-side tool definition and the
|
||||
response-side tool_use block."""
|
||||
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
||||
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
||||
if not base_url or not api_key:
|
||||
|
|
@ -105,10 +130,11 @@ def test_web_search_vertex_ai(compat_result):
|
|||
failures.append(error)
|
||||
continue
|
||||
|
||||
if not _has_server_tool_block(outcome.events):
|
||||
if not _has_web_search_tool_use(outcome.events):
|
||||
error = (
|
||||
f"[{model}] no server_tool_use / web_search_tool_result block "
|
||||
"observed"
|
||||
f"[{model}] no `tool_use` block with name=WebSearch observed; "
|
||||
"the proxy may have stripped the WebSearch tool definition from "
|
||||
"the request or the tool_use block from the response"
|
||||
)
|
||||
compat_result.add({"status": "fail", "error": error})
|
||||
failures.append(error)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue