mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-25 01:02:15 +00:00
fix(anthropic): forward Claude Code safeguards and dangerous-tool-use beta to Bedrock Invoke and Vertex on /v1/messages
Claude Code's server-side auto-mode classifier sends a `safeguards` body field together with the `dangerous-tool-use-2026-09-03` beta. PR #42152 made the first-party anthropic route pass them through, but the beta header mapping left the other two Claude platforms at null, so Bedrock Invoke dropped both (classifier silently disabled) and Vertex forwarded the body field without the beta, which the platform rejects with "safeguards: Extra inputs are not permitted" (a 400 Claude Code hides by retrying without them). Map the beta for bedrock and vertex_ai in the beta headers config and add `safeguards` to the Bedrock Invoke request allowlist so the pair reaches both platforms unchanged. Nothing is injected: a client that sends `safeguards` without the beta still gets the platform's 400, exactly as api.anthropic.com answers it.
This commit is contained in:
parent
24f0f373fc
commit
0e16050100
5 changed files with 159 additions and 0 deletions
|
|
@ -11,6 +11,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": "fast-mode-2026-02-01",
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
@ -44,6 +45,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
@ -76,6 +78,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": null,
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -109,6 +112,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -142,6 +146,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": null,
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -175,6 +180,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": "fast-mode-2026-02-01",
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
|
|||
|
|
@ -1236,6 +1236,7 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
|
|||
thinking: dict
|
||||
metadata: dict
|
||||
output_config: dict
|
||||
safeguards: list
|
||||
|
||||
# `context_management` is allowed for Bedrock InvokeModel only when it
|
||||
# carries `compact_20260112` edits paired with the `compact-2026-01-12`
|
||||
|
|
|
|||
|
|
@ -1544,3 +1544,98 @@ async def test_anthropic_messages_streaming_forwards_safeguards_and_keeps_safegu
|
|||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert events[0]["message"]["safeguard_results"] == safeguard_results
|
||||
assert [e for e in events if e["type"] == "message_delta"][0]["delta"]["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
def _claude_code_auto_mode_request() -> tuple[list[dict[str, object]], list[dict[str, object]]]:
|
||||
"""Shapes are what Claude Code 2.1.278 sends and Bedrock Invoke / Vertex rawPredict return, captured 2026-09-21."""
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
return safeguards, safeguard_results
|
||||
|
||||
|
||||
def _upstream_answering_with(safeguard_results: list[dict[str, object]], captured: dict[str, object]) -> AsyncHTTPHandler:
|
||||
def upstream_records_the_request(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
captured["anthropic-beta"] = request.headers.get("anthropic-beta")
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-sonnet-5",
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
request=request,
|
||||
)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_records_the_request))
|
||||
return upstream
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_forwards_safeguards_and_dangerous_tool_use_beta_to_bedrock_invoke(
|
||||
local_beta_headers_config,
|
||||
):
|
||||
"""Bedrock Invoke takes betas in the body's `anthropic_beta` and 400s on `safeguards` without the beta."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards, safeguard_results = _claude_code_auto_mode_request()
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="bedrock/us.anthropic.claude-sonnet-5",
|
||||
custom_llm_provider="bedrock",
|
||||
aws_access_key_id="test-access-key",
|
||||
aws_secret_access_key="test-secret-key",
|
||||
aws_region_name="us-east-1",
|
||||
client=_upstream_answering_with(safeguard_results, captured),
|
||||
safeguards=safeguards,
|
||||
extra_headers={"anthropic-beta": "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"},
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert captured["body"]["anthropic_beta"] == ["dangerous-tool-use-2026-09-03"]
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_forwards_safeguards_and_dangerous_tool_use_beta_to_vertex(
|
||||
local_beta_headers_config,
|
||||
):
|
||||
"""Vertex rawPredict takes the beta as the `anthropic-beta` header and 400s on `safeguards` without it."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
|
||||
|
||||
safeguards, safeguard_results = _claude_code_auto_mode_request()
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
with patch.object(VertexBase, "_ensure_access_token", return_value=("test-token", "test-project")):
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="vertex_ai/claude-sonnet-5",
|
||||
custom_llm_provider="vertex_ai",
|
||||
vertex_project="test-project",
|
||||
vertex_location="global",
|
||||
vertex_credentials="{}",
|
||||
client=_upstream_answering_with(safeguard_results, captured),
|
||||
safeguards=safeguards,
|
||||
extra_headers={"anthropic-beta": "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"},
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert "anthropic_beta" not in captured["body"]
|
||||
assert set(captured["anthropic-beta"].split(",")) == {
|
||||
"dangerous-tool-use-2026-09-03",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
}
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
|
|
|||
|
|
@ -1651,6 +1651,49 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
|
|||
assert set(result).issubset(cfg.BEDROCK_INVOKE_ALLOWED_TOP_LEVEL_FIELDS)
|
||||
|
||||
|
||||
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(local_beta_headers_config):
|
||||
"""
|
||||
Claude Code's server-side auto-mode classifier sends `safeguards` alongside the
|
||||
dangerous-tool-use-2026-09-03 beta. Bedrock Invoke accepts the pair, answers
|
||||
"safeguards: Extra inputs are not permitted" for the field alone, and returns
|
||||
`safeguard_results: []` for the beta alone, so both must reach it unchanged.
|
||||
"""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model="us.anthropic.claude-sonnet-5",
|
||||
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
|
||||
anthropic_messages_optional_request_params={"max_tokens": 64, "safeguards": safeguards},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"anthropic-beta": "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"},
|
||||
)
|
||||
|
||||
assert result["safeguards"] == safeguards
|
||||
assert result["anthropic_beta"] == ["dangerous-tool-use-2026-09-03"]
|
||||
|
||||
|
||||
def test_bedrock_messages_stream_decoder_keeps_safeguard_results():
|
||||
"""Bedrock streams the classifier verdicts on message_start and on the final message_delta, exactly as api.anthropic.com does."""
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="us.anthropic.claude-sonnet-5")
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
|
||||
message_delta = decoder._chunk_parser(
|
||||
{
|
||||
"type": "message_delta",
|
||||
"delta": {"stop_reason": "end_turn", "stop_sequence": None, "safeguard_results": safeguard_results},
|
||||
"usage": {"output_tokens": 1},
|
||||
"amazon-bedrock-invocationMetrics": {"inputTokenCount": 3, "outputTokenCount": 1},
|
||||
}
|
||||
)
|
||||
|
||||
assert isinstance(message_delta, dict)
|
||||
assert message_delta["delta"]["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
def test_bedrock_messages_filters_user_provided_unsupported_beta_header():
|
||||
"""
|
||||
In proxy deployments the client (e.g. Claude Code) doesn't know the backend
|
||||
|
|
|
|||
|
|
@ -442,6 +442,20 @@ class TestAnthropicBetaHeadersFiltering:
|
|||
|
||||
assert filtered == ["thinking-binding-controls-2026-08-01"]
|
||||
|
||||
@pytest.mark.parametrize("provider", ["anthropic", "bedrock", "vertex_ai"])
|
||||
def test_dangerous_tool_use_forwarded(self, provider):
|
||||
"""Claude Code's server-side auto-mode classifier sends `safeguards` together with
|
||||
dangerous-tool-use-2026-09-03. Bedrock Invoke and Vertex rawPredict both answer
|
||||
"safeguards: Extra inputs are not permitted" when the body field arrives without
|
||||
the beta (probed 2026-09-21), so dropping the header turned every auto-mode turn
|
||||
into a 400 on Vertex and silently disabled the classifier on Bedrock."""
|
||||
filtered = filter_and_transform_beta_headers(
|
||||
beta_headers=["dangerous-tool-use-2026-09-03"],
|
||||
provider=provider,
|
||||
)
|
||||
|
||||
assert filtered == ["dangerous-tool-use-2026-09-03"]
|
||||
|
||||
def test_null_value_headers_filtered(self):
|
||||
"""Test that headers with null values are always filtered out."""
|
||||
for provider in [
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue