mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
Merge pull request #42538 from BerriAI/litellm_cherrypick_safeguards_1_102_x
fix(anthropic): backport #42152 and #42288 to stable/1.102.x for v1.102.1
This commit is contained in:
commit
89000e402f
18 changed files with 436 additions and 9 deletions
|
|
@ -11,6 +11,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": "fast-mode-2026-02-01",
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
@ -42,6 +43,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
@ -72,6 +74,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": null,
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -103,6 +106,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -134,6 +138,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": "dangerous-tool-use-2026-09-03",
|
||||
"effort-2025-11-24": null,
|
||||
"fast-mode-2026-02-01": null,
|
||||
"files-api-2025-04-14": null,
|
||||
|
|
@ -165,6 +170,7 @@
|
|||
"computer-use-2025-11-24": "computer-use-2025-11-24",
|
||||
"context-1m-2025-08-07": "context-1m-2025-08-07",
|
||||
"context-management-2025-06-27": "context-management-2025-06-27",
|
||||
"dangerous-tool-use-2026-09-03": null,
|
||||
"effort-2025-11-24": "effort-2025-11-24",
|
||||
"fast-mode-2026-02-01": "fast-mode-2026-02-01",
|
||||
"files-api-2025-04-14": "files-api-2025-04-14",
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ if TYPE_CHECKING:
|
|||
from litellm.router import Router
|
||||
|
||||
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
|
||||
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config"})
|
||||
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config", "safeguards"})
|
||||
|
||||
_AnthropicMessages: TypeAlias = "list[dict[str, object]]"
|
||||
_AnthropicSystem: TypeAlias = "str | list[dict[str, object]] | None"
|
||||
|
|
|
|||
|
|
@ -70,10 +70,14 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
"speed",
|
||||
"output_config",
|
||||
"reasoning_effort",
|
||||
"safeguards",
|
||||
# TODO: Add Anthropic `metadata` support
|
||||
# "metadata",
|
||||
]
|
||||
|
||||
def should_filter_anthropic_beta_headers(self) -> bool:
|
||||
return self._resolved_provider != "anthropic"
|
||||
|
||||
def _remove_scope_from_cache_control(self, anthropic_messages_request: dict) -> None:
|
||||
"""
|
||||
Remove `scope` field from cache_control blocks.
|
||||
|
|
|
|||
|
|
@ -561,6 +561,9 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
if injected_thinking_for_clear_thinking:
|
||||
beta_set.add("interleaved-thinking-2025-05-14")
|
||||
|
||||
if anthropic_messages_optional_request_params.get("safeguards") is not None:
|
||||
beta_set.add(ANTHROPIC_BETA_HEADER_VALUES.DANGEROUS_TOOL_USE_2026_09_03.value)
|
||||
|
||||
self._filter_context_management_for_bedrock_invoke(
|
||||
anthropic_messages_request=anthropic_messages_request,
|
||||
beta_set=beta_set,
|
||||
|
|
|
|||
|
|
@ -108,6 +108,9 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
|
|||
if anthropic_model_info.is_tool_search_used(tools):
|
||||
beta_values.add(get_tool_search_beta_header("vertex_ai"))
|
||||
|
||||
if optional_params.get("safeguards") is not None:
|
||||
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.DANGEROUS_TOOL_USE_2026_09_03.value)
|
||||
|
||||
if beta_values:
|
||||
headers["anthropic-beta"] = ",".join(beta_values)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
from collections.abc import Iterable, Sequence
|
||||
from collections.abc import Iterable, Mapping, Sequence
|
||||
from enum import Enum
|
||||
from typing import Any, Final, Literal, TypeAlias
|
||||
|
||||
|
|
@ -410,6 +410,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
|
|||
output_config: AnthropicOutputConfig | None # Configuration for Claude's output behavior
|
||||
cache_control: dict[str, Any] | None # Automatic prompt caching
|
||||
reasoning_effort: str | None
|
||||
safeguards: ReadOnly[Sequence[Mapping[str, object]] | None]
|
||||
|
||||
|
||||
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):
|
||||
|
|
@ -529,6 +530,7 @@ class AnthropicStopDetails(TypedDict, total=False):
|
|||
class MessageDelta(TypedDict, total=False):
|
||||
stop_reason: str | None
|
||||
stop_details: ReadOnly[AnthropicStopDetails]
|
||||
safeguard_results: ReadOnly[Sequence[Mapping[str, object]]]
|
||||
|
||||
|
||||
class ServerToolUsage(TypedDict, total=False):
|
||||
|
|
@ -599,6 +601,7 @@ class MessageChunk(TypedDict, total=False):
|
|||
stop_reason: str | None
|
||||
stop_sequence: str | None
|
||||
usage: UsageDelta
|
||||
safeguard_results: ReadOnly[Sequence[Mapping[str, object]]]
|
||||
|
||||
|
||||
class MessageStartBlock(TypedDict):
|
||||
|
|
@ -746,6 +749,7 @@ class ANTHROPIC_BETA_HEADER_VALUES(str, Enum):
|
|||
ADVANCED_TOOL_USE_2025_11_20 = "advanced-tool-use-2025-11-20"
|
||||
FAST_MODE_2026_02_01 = "fast-mode-2026-02-01"
|
||||
ADVISOR_TOOL_2026_03_01 = "advisor-tool-2026-03-01"
|
||||
DANGEROUS_TOOL_USE_2026_09_03 = "dangerous-tool-use-2026-09-03"
|
||||
|
||||
|
||||
# Tool search beta header constant (for Anthropic direct API and Microsoft Foundry)
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from typing import Any, Literal, TypeAlias
|
||||
|
||||
from typing_extensions import NotRequired, ReadOnly, TypedDict
|
||||
|
|
@ -97,3 +98,4 @@ class AnthropicMessagesResponse(TypedDict, total=False):
|
|||
type: Literal["message"] | None
|
||||
usage: AnthropicUsage | None
|
||||
context_management: NotRequired[ContextManagementResponse]
|
||||
safeguard_results: NotRequired[ReadOnly[Sequence[Mapping[str, object]]]]
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import json
|
||||
from collections.abc import Sequence
|
||||
from collections.abc import Mapping, Sequence
|
||||
from enum import Enum
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias
|
||||
|
||||
|
|
@ -1215,6 +1215,7 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
|
|||
thinking: dict
|
||||
metadata: dict
|
||||
output_config: dict
|
||||
safeguards: ReadOnly[Sequence[Mapping[str, object]]]
|
||||
|
||||
# `context_management` is allowed for Bedrock InvokeModel only when it
|
||||
# carries `compact_20260112` edits paired with the `compact-2026-01-12`
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[project]
|
||||
name = "litellm"
|
||||
version = "1.102.0"
|
||||
version = "1.102.1"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10, <3.15"
|
||||
|
|
@ -330,7 +330,7 @@ members = ["enterprise", "litellm-proxy-extras"]
|
|||
profile = "black"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "1.102.0"
|
||||
version = "1.102.1"
|
||||
version_files = [
|
||||
"pyproject.toml:^version",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -101,8 +101,8 @@ async def test_anthropic_messages_with_all_beta_headers(model_name, provider_nam
|
|||
@pytest.mark.parametrize(
|
||||
"model_name,provider_name",
|
||||
[
|
||||
("bedrock-claude-opus-4.5", "bedrock"),
|
||||
("bedrock-converse-claude-sonnet-4.5", "bedrock_converse"),
|
||||
("bedrock-claude-fable-5.1", "bedrock"),
|
||||
("bedrock-converse-claude-fable-5.1", "bedrock_converse"),
|
||||
],
|
||||
)
|
||||
async def test_bedrock_invoke_messages_with_all_beta_headers(model_name, provider_name):
|
||||
|
|
|
|||
|
|
@ -25,6 +25,11 @@ model_list:
|
|||
model: "bedrock/us.anthropic.claude-opus-4-5-20251101-v1:0"
|
||||
aws_region_name: "us-east-1"
|
||||
|
||||
- model_name: bedrock-claude-fable-5.1
|
||||
litellm_params:
|
||||
model: "bedrock/us.anthropic.claude-fable-5-1"
|
||||
aws_region_name: "us-east-1"
|
||||
|
||||
- model_name: bedrock-nova-pro
|
||||
litellm_params:
|
||||
model: "bedrock/us.amazon.nova-pro-v1:0"
|
||||
|
|
@ -35,6 +40,11 @@ model_list:
|
|||
litellm_params:
|
||||
model: "bedrock/converse/us.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||
aws_region_name: "us-east-1"
|
||||
|
||||
- model_name: bedrock-converse-claude-fable-5.1
|
||||
litellm_params:
|
||||
model: "bedrock/converse/us.anthropic.claude-fable-5-1"
|
||||
aws_region_name: "us-east-1"
|
||||
|
||||
# Azure AI models
|
||||
- model_name: azure-ai-claude-opus-4.5
|
||||
|
|
|
|||
|
|
@ -206,6 +206,21 @@ def local_model_cost_map(monkeypatch):
|
|||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_beta_headers_config(monkeypatch):
|
||||
"""Pin the bundled ``anthropic_beta_headers_config.json`` so beta header assertions
|
||||
do not depend on the network-fetched copy or on what earlier tests left cached."""
|
||||
from litellm.anthropic_beta_headers_manager import reload_beta_headers_config
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS", "True")
|
||||
reload_beta_headers_config()
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
monkeypatch.delenv("LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS", raising=False)
|
||||
reload_beta_headers_config()
|
||||
|
||||
|
||||
def _run_coroutine_if_needed(result):
|
||||
if not asyncio.iscoroutine(result):
|
||||
return
|
||||
|
|
|
|||
|
|
@ -110,6 +110,19 @@ class TestOutputConfigStrippedFromCompletionKwargs:
|
|||
"reject it with 400 'Extra inputs are not permitted'"
|
||||
)
|
||||
|
||||
def test_safeguards_is_stripped_for_non_anthropic_target(self):
|
||||
extra_kwargs = {
|
||||
"custom_llm_provider": "azure",
|
||||
"safeguards": [{"type": "dangerous_tool_use", "classifier_context": {"v": 1}}],
|
||||
}
|
||||
|
||||
result = _call_prepare(extra_kwargs=extra_kwargs)
|
||||
|
||||
completion_kwargs = result[0] if isinstance(result, tuple) else result
|
||||
assert "safeguards" not in completion_kwargs, (
|
||||
"safeguards is an Anthropic-only field; OpenAI-format backends reject it with 400"
|
||||
)
|
||||
|
||||
def test_output_config_format_translated_to_response_format(self):
|
||||
"""When ``output_config`` carries structured-output ``format``, the
|
||||
translator now maps it to OpenAI's ``response_format`` so non-Anthropic
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@ import asyncio
|
|||
import json
|
||||
import os
|
||||
import uuid
|
||||
from typing import Any, Dict, List
|
||||
from typing import Any, Dict, Final, List
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
|
@ -1438,3 +1438,212 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped():
|
|||
)
|
||||
|
||||
assert "Traceback" not in str(excinfo.value)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_forwards_safeguards_and_unknown_beta_to_anthropic():
|
||||
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
client_betas = "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": {}}}]
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
def upstream_records_the_request(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
captured["anthropic-beta"] = request.headers.get("anthropic-beta")
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-haiku-4-5",
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
request=request,
|
||||
)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_records_the_request))
|
||||
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="anthropic/claude-haiku-4-5",
|
||||
custom_llm_provider="anthropic",
|
||||
api_key="sk-test",
|
||||
client=upstream,
|
||||
safeguards=safeguards,
|
||||
extra_headers={"anthropic-beta": client_betas},
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert set(captured["anthropic-beta"].split(",")) == set(client_betas.split(","))
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_streaming_forwards_safeguards_and_keeps_safeguard_results():
|
||||
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
captured: dict[str, object] = {}
|
||||
message_start = {
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-haiku-4-5",
|
||||
"content": [],
|
||||
"stop_reason": None,
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 0},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
}
|
||||
message_delta = {
|
||||
"type": "message_delta",
|
||||
"delta": {"stop_reason": "end_turn", "stop_sequence": None, "safeguard_results": safeguard_results},
|
||||
"usage": {"output_tokens": 1},
|
||||
}
|
||||
sse = "".join(
|
||||
f"event: {event['type']}\ndata: {json.dumps(event)}\n\n"
|
||||
for event in (message_start, message_delta, {"type": "message_stop"})
|
||||
)
|
||||
|
||||
def upstream_streams_safeguard_results(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse.encode(), request=request)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_streams_safeguard_results))
|
||||
|
||||
stream = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="anthropic/claude-haiku-4-5",
|
||||
custom_llm_provider="anthropic",
|
||||
api_key="sk-test",
|
||||
client=upstream,
|
||||
stream=True,
|
||||
safeguards=safeguards,
|
||||
)
|
||||
raw = b"".join([chunk async for chunk in stream]).decode()
|
||||
events = [json.loads(line[len("data: ") :]) for line in raw.splitlines() if line.startswith("data: ")]
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert events[0]["message"]["safeguard_results"] == safeguard_results
|
||||
assert [e for e in events if e["type"] == "message_delta"][0]["delta"]["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
def _claude_code_auto_mode_request() -> tuple[list[dict[str, object]], list[dict[str, object]]]:
|
||||
"""Shapes are what Claude Code 2.1.278 sends and Bedrock Invoke / Vertex rawPredict return, captured 2026-09-21."""
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
return safeguards, safeguard_results
|
||||
|
||||
|
||||
def _upstream_answering_with(safeguard_results: list[dict[str, object]], captured: dict[str, object]) -> AsyncHTTPHandler:
|
||||
def upstream_records_the_request(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
captured["anthropic-beta"] = request.headers.get("anthropic-beta")
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-sonnet-5",
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
request=request,
|
||||
)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_records_the_request))
|
||||
return upstream
|
||||
|
||||
|
||||
_CLIENT_BETA_HEADERS: Final = (
|
||||
pytest.param({"anthropic-beta": "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"}, id="client_sends_beta"),
|
||||
pytest.param({"anthropic-beta": "interleaved-thinking-2025-05-14"}, id="client_omits_beta"),
|
||||
pytest.param({}, id="client_sends_no_beta_header"),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("client_headers", _CLIENT_BETA_HEADERS)
|
||||
async def test_anthropic_messages_forwards_safeguards_and_dangerous_tool_use_beta_to_bedrock_invoke(
|
||||
local_beta_headers_config, client_headers
|
||||
):
|
||||
"""Bedrock Invoke takes betas in the body's `anthropic_beta` and 400s on `safeguards` without the beta, so the beta rides along with the field."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards, safeguard_results = _claude_code_auto_mode_request()
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="bedrock/us.anthropic.claude-sonnet-5",
|
||||
custom_llm_provider="bedrock",
|
||||
aws_access_key_id="test-access-key",
|
||||
aws_secret_access_key="test-secret-key",
|
||||
aws_region_name="us-east-1",
|
||||
client=_upstream_answering_with(safeguard_results, captured),
|
||||
safeguards=safeguards,
|
||||
extra_headers=client_headers,
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert captured["body"]["anthropic_beta"] == ["dangerous-tool-use-2026-09-03"]
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("client_headers", _CLIENT_BETA_HEADERS)
|
||||
async def test_anthropic_messages_forwards_safeguards_and_dangerous_tool_use_beta_to_vertex(
|
||||
local_beta_headers_config, client_headers
|
||||
):
|
||||
"""Vertex rawPredict takes the beta as the `anthropic-beta` header and 400s on `safeguards` without it, so the beta rides along with the field."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
|
||||
|
||||
safeguards, safeguard_results = _claude_code_auto_mode_request()
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
with patch.object( # test-quality-ok: the GCP token exchange runs before the faked HTTP boundary
|
||||
VertexBase, "_ensure_access_token", return_value=("test-token", "test-project")
|
||||
):
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="vertex_ai/claude-sonnet-5",
|
||||
custom_llm_provider="vertex_ai",
|
||||
vertex_project="test-project",
|
||||
vertex_location="global",
|
||||
vertex_credentials="{}",
|
||||
client=_upstream_answering_with(safeguard_results, captured),
|
||||
safeguards=safeguards,
|
||||
extra_headers=client_headers,
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert "anthropic_beta" not in captured["body"]
|
||||
assert captured["anthropic-beta"].split(",").count("dangerous-tool-use-2026-09-03") == 1
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
|
|
|||
|
|
@ -1648,6 +1648,92 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
|
|||
assert set(result).issubset(cfg.BEDROCK_INVOKE_ALLOWED_TOP_LEVEL_FIELDS)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"client_beta_header",
|
||||
["dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14", "interleaved-thinking-2025-05-14"],
|
||||
ids=["client_sends_beta", "client_omits_beta"],
|
||||
)
|
||||
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(local_beta_headers_config, client_beta_header):
|
||||
"""
|
||||
Claude Code's server-side auto-mode classifier sends `safeguards` alongside the
|
||||
dangerous-tool-use-2026-09-03 beta. Bedrock Invoke accepts the pair, answers
|
||||
"safeguards: Extra inputs are not permitted" for the field alone, and returns
|
||||
`safeguard_results: []` for the beta alone, so the field reaches it unchanged
|
||||
and the beta rides along whether or not the client sent it, as every other
|
||||
body-driven beta does here.
|
||||
"""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model="us.anthropic.claude-sonnet-5",
|
||||
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
|
||||
anthropic_messages_optional_request_params={"max_tokens": 64, "safeguards": safeguards},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"anthropic-beta": client_beta_header},
|
||||
)
|
||||
|
||||
assert result["safeguards"] == safeguards
|
||||
assert result["anthropic_beta"].count("dangerous-tool-use-2026-09-03") == 1
|
||||
|
||||
|
||||
def test_bedrock_messages_does_not_add_dangerous_tool_use_beta_without_safeguards(local_beta_headers_config):
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model="us.anthropic.claude-sonnet-5",
|
||||
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
|
||||
anthropic_messages_optional_request_params={"max_tokens": 64},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "safeguards" not in result
|
||||
assert "dangerous-tool-use-2026-09-03" not in result.get("anthropic_beta", [])
|
||||
|
||||
|
||||
def test_bedrock_messages_stream_decoder_keeps_safeguard_results():
|
||||
"""Bedrock streams the classifier verdicts on message_start and on the final message_delta, exactly as api.anthropic.com does."""
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="us.anthropic.claude-sonnet-5")
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
|
||||
message_start = decoder._chunk_parser(
|
||||
{
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"id": "msg_01",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-sonnet-5",
|
||||
"content": [],
|
||||
"stop_reason": None,
|
||||
"usage": {"input_tokens": 3, "output_tokens": 0},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
assert isinstance(message_start, dict)
|
||||
assert message_start["message"]["safeguard_results"] == safeguard_results
|
||||
|
||||
message_delta = decoder._chunk_parser(
|
||||
{
|
||||
"type": "message_delta",
|
||||
"delta": {"stop_reason": "end_turn", "stop_sequence": None, "safeguard_results": safeguard_results},
|
||||
"usage": {"output_tokens": 1},
|
||||
"amazon-bedrock-invocationMetrics": {"inputTokenCount": 3, "outputTokenCount": 1},
|
||||
}
|
||||
)
|
||||
|
||||
assert isinstance(message_delta, dict)
|
||||
assert message_delta["delta"]["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
def test_bedrock_messages_filters_user_provided_unsupported_beta_header():
|
||||
"""
|
||||
In proxy deployments the client (e.g. Claude Code) doesn't know the backend
|
||||
|
|
|
|||
|
|
@ -81,6 +81,63 @@ def test_web_search_header_added_for_messages_endpoint():
|
|||
), f"anthropic-beta should be 'web-search-2025-03-05', got: {updated_headers['anthropic-beta']}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"client_headers",
|
||||
[{"anthropic-beta": "dangerous-tool-use-2026-09-03"}, {}],
|
||||
ids=["client_sends_beta", "client_omits_beta"],
|
||||
)
|
||||
def test_safeguards_add_dangerous_tool_use_beta_header(client_headers):
|
||||
"""Vertex rejects `safeguards` without the dangerous-tool-use beta, so the beta rides along with the field the way the web search and context management betas do."""
|
||||
config = VertexAIPartnerModelsAnthropicMessagesConfig()
|
||||
litellm_params = {
|
||||
"vertex_ai_project": "test-project",
|
||||
"vertex_ai_location": "global",
|
||||
"vertex_credentials": "{}",
|
||||
}
|
||||
optional_params = {
|
||||
"safeguards": [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
}
|
||||
|
||||
with (
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers=client_headers,
|
||||
model="claude-sonnet-5",
|
||||
messages=[],
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert updated_headers["anthropic-beta"].split(",").count("dangerous-tool-use-2026-09-03") == 1
|
||||
|
||||
|
||||
def test_no_safeguards_leaves_dangerous_tool_use_beta_header_out():
|
||||
config = VertexAIPartnerModelsAnthropicMessagesConfig()
|
||||
litellm_params = {
|
||||
"vertex_ai_project": "test-project",
|
||||
"vertex_ai_location": "global",
|
||||
"vertex_credentials": "{}",
|
||||
}
|
||||
|
||||
with (
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, _ = config.validate_anthropic_messages_environment(
|
||||
headers={},
|
||||
model="claude-sonnet-5",
|
||||
messages=[],
|
||||
optional_params={"max_tokens": 64},
|
||||
litellm_params=litellm_params,
|
||||
api_base=None,
|
||||
)
|
||||
|
||||
assert "dangerous-tool-use-2026-09-03" not in updated_headers.get("anthropic-beta", "")
|
||||
|
||||
|
||||
def test_web_search_header_not_added_without_tool():
|
||||
"""Test that beta header is NOT added when web search tool is not present"""
|
||||
config = VertexAIPartnerModelsAnthropicMessagesConfig()
|
||||
|
|
|
|||
|
|
@ -426,6 +426,20 @@ class TestAnthropicBetaHeadersFiltering:
|
|||
|
||||
assert filtered == ["fine-grained-tool-streaming-2025-05-14"]
|
||||
|
||||
@pytest.mark.parametrize("provider", ["anthropic", "bedrock", "vertex_ai"])
|
||||
def test_dangerous_tool_use_forwarded(self, provider):
|
||||
"""Claude Code's server-side auto-mode classifier sends `safeguards` together with
|
||||
dangerous-tool-use-2026-09-03. Bedrock Invoke and Vertex rawPredict both answer
|
||||
"safeguards: Extra inputs are not permitted" when the body field arrives without
|
||||
the beta (probed 2026-09-21), so dropping the header turned every auto-mode turn
|
||||
into a 400 on Vertex and silently disabled the classifier on Bedrock."""
|
||||
filtered = filter_and_transform_beta_headers(
|
||||
beta_headers=["dangerous-tool-use-2026-09-03"],
|
||||
provider=provider,
|
||||
)
|
||||
|
||||
assert filtered == ["dangerous-tool-use-2026-09-03"]
|
||||
|
||||
def test_null_value_headers_filtered(self):
|
||||
"""Test that headers with null values are always filtered out."""
|
||||
for provider in [
|
||||
|
|
|
|||
2
uv.lock
generated
2
uv.lock
generated
|
|
@ -4358,7 +4358,7 @@ wheels = [
|
|||
|
||||
[[package]]
|
||||
name = "litellm"
|
||||
version = "1.102.0"
|
||||
version = "1.102.1"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "aiohttp" },
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue