mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
fix(bedrock): add beta for thinking display updates (#43832)
This commit is contained in:
parent
4b06d04334
commit
c2ae483782
9 changed files with 395 additions and 225 deletions
|
|
@ -34,9 +34,11 @@ from litellm.types.llms.anthropic import (
|
|||
ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER,
|
||||
ANTHROPIC_OAUTH_BETA_HEADER,
|
||||
ANTHROPIC_OAUTH_TOKEN_PREFIX,
|
||||
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,
|
||||
AllAnthropicToolsValues,
|
||||
AnthropicMcpServerTool,
|
||||
AnthropicMessagesToolChoice,
|
||||
AnthropicThinkingParam,
|
||||
)
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.proxy.model_listing import ModelInfoResponse
|
||||
|
|
@ -337,6 +339,11 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
file_ids: Final = get_file_ids_from_messages(messages)
|
||||
return len(file_ids) > 0
|
||||
|
||||
def is_thinking_display_updates_used(self, thinking: AnthropicThinkingParam | None) -> bool:
|
||||
if not isinstance(thinking, dict):
|
||||
return False
|
||||
return thinking.get("type") in ("adaptive", "enabled") and thinking.get("display") == "updates"
|
||||
|
||||
def is_mid_conversation_output_config_used(self, messages: list[AllMessageValues]) -> bool:
|
||||
"""
|
||||
Return if "output_config" is in a message
|
||||
|
|
@ -749,7 +756,11 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
existing_output_config: Final = optional_params.get("output_config")
|
||||
optional_params["thinking"] = {"type": "adaptive"}
|
||||
display: Final = thinking.get("display")
|
||||
if display in ("summarized", "omitted"):
|
||||
optional_params["thinking"] = {"type": "adaptive", "display": display}
|
||||
else:
|
||||
optional_params["thinking"] = {"type": "adaptive"}
|
||||
optional_params["output_config"] = {
|
||||
"effort": effort,
|
||||
**(existing_output_config if isinstance(existing_output_config, dict) else MappingProxyType({})),
|
||||
|
|
@ -869,6 +880,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
*,
|
||||
custom_llm_provider: str,
|
||||
is_mid_conversation_output_config_used: bool = False,
|
||||
is_thinking_display_updates_used: bool = False,
|
||||
) -> list[str]:
|
||||
"""
|
||||
Get list of common beta headers based on the features that are active.
|
||||
|
|
@ -904,7 +916,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
if is_mid_conversation_output_config_used:
|
||||
betas.append(ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER)
|
||||
|
||||
return list(set(betas))
|
||||
thinking_display_betas: Final = (
|
||||
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,) if is_thinking_display_updates_used else ()
|
||||
)
|
||||
return list(set(betas).union(thinking_display_betas))
|
||||
|
||||
@staticmethod
|
||||
def _make_api_key_auth_header(api_key: str, api_base: str | None, use_bearer_for_custom_base: bool = False) -> dict:
|
||||
|
|
@ -937,6 +952,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
api_base: str | None = None,
|
||||
use_bearer_for_custom_base: bool = False,
|
||||
is_mid_conversation_output_config_used: bool = False,
|
||||
is_thinking_display_updates_used: bool = False,
|
||||
) -> dict:
|
||||
betas: Final = set()
|
||||
# Anthropic no longer requires the prompt-caching beta header
|
||||
|
|
@ -993,6 +1009,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
if user_anthropic_beta_headers is not None:
|
||||
betas.update(user_anthropic_beta_headers)
|
||||
|
||||
all_betas: Final = betas.union(
|
||||
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,) if is_thinking_display_updates_used else ()
|
||||
)
|
||||
|
||||
# Don't send any beta headers to Vertex, except web search which is required
|
||||
if is_vertex_request is True:
|
||||
# Vertex AI requires web search beta header for web search to work
|
||||
|
|
@ -1000,8 +1020,8 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
|
||||
|
||||
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
|
||||
elif len(betas) > 0:
|
||||
headers["anthropic-beta"] = ",".join(betas)
|
||||
elif len(all_betas) > 0:
|
||||
headers["anthropic-beta"] = ",".join(all_betas)
|
||||
|
||||
return headers
|
||||
|
||||
|
|
@ -1059,6 +1079,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
auth_token=auth_token,
|
||||
file_id_used=file_id_used,
|
||||
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
|
||||
is_thinking_display_updates_used=self.is_thinking_display_updates_used(optional_params.get("thinking")),
|
||||
web_search_tool_used=web_search_tool_used,
|
||||
is_vertex_request=optional_params.get("is_vertex_request", False),
|
||||
user_anthropic_beta_headers=user_anthropic_beta_headers,
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
|
|||
from litellm.types.llms.anthropic import (
|
||||
ANTHROPIC_ADVISOR_TOOL_TYPE,
|
||||
ANTHROPIC_BETA_HEADER_VALUES,
|
||||
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,
|
||||
AnthropicMessagesRequest,
|
||||
)
|
||||
from litellm.types.llms.anthropic_messages.anthropic_response import (
|
||||
|
|
@ -688,8 +689,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
if AnthropicModelInfo().is_tool_search_used(tools):
|
||||
beta_values.add(get_tool_search_beta_header(custom_llm_provider))
|
||||
|
||||
if not beta_values:
|
||||
thinking_display_betas: Final = (
|
||||
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,)
|
||||
if AnthropicModelInfo().is_thinking_display_updates_used(optional_params.get("thinking"))
|
||||
else ()
|
||||
)
|
||||
all_beta_values: Final = beta_values.union(thinking_display_betas)
|
||||
|
||||
if not all_beta_values:
|
||||
return headers
|
||||
merged: Final = {key: value for key, value in headers.items() if key.lower() != "anthropic-beta"}
|
||||
merged["anthropic-beta"] = ",".join(sorted(beta_values))
|
||||
merged["anthropic-beta"] = ",".join(sorted(all_beta_values))
|
||||
return merged
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from litellm.llms.bedrock.common_utils import (
|
|||
from litellm.types.llms.anthropic import (
|
||||
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER,
|
||||
ANTHROPIC_TOOL_SEARCH_BETA_HEADER,
|
||||
AnthropicThinkingParam,
|
||||
)
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
|
@ -85,6 +86,8 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
from litellm.utils import supports_native_structured_output
|
||||
|
||||
original_model: Final = model
|
||||
requested_thinking: Final = non_default_params.get("thinking")
|
||||
requested_display_updates: Final = self.is_thinking_display_updates_used(requested_thinking)
|
||||
if "response_format" in non_default_params and not supports_native_structured_output(
|
||||
model=model, custom_llm_provider="bedrock"
|
||||
):
|
||||
|
|
@ -114,6 +117,14 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model(
|
||||
model=original_model, optional_params=optional_params, custom_llm_provider="bedrock"
|
||||
)
|
||||
translated_thinking: Final = optional_params.get("thinking")
|
||||
if (
|
||||
requested_display_updates
|
||||
and isinstance(translated_thinking, dict)
|
||||
and translated_thinking.get("type") == "adaptive"
|
||||
):
|
||||
thinking_with_display: Final[AnthropicThinkingParam] = {"type": "adaptive", "display": "updates"}
|
||||
optional_params["thinking"] = thinking_with_display
|
||||
|
||||
# The stub model hides the original model from the parent's forced-tool-use backstop
|
||||
response_format_tool_choice: Final = optional_params.get("tool_choice")
|
||||
|
|
@ -170,6 +181,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
headers=headers,
|
||||
thinking=_anthropic_request.get("thinking"),
|
||||
)
|
||||
if beta_list:
|
||||
_anthropic_request["anthropic_beta"] = beta_list
|
||||
|
|
@ -250,12 +262,14 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
messages: list[AllMessageValues],
|
||||
optional_params: dict,
|
||||
headers: dict,
|
||||
thinking: AnthropicThinkingParam | None,
|
||||
) -> list[str]:
|
||||
tools: Final = optional_params.get("tools")
|
||||
tool_search_used: Final = self.is_tool_search_used(tools)
|
||||
programmatic_tool_calling_used: Final = self.is_programmatic_tool_calling_used(tools)
|
||||
input_examples_used: Final = self.is_input_examples_used(tools)
|
||||
is_mid_conversation_output_config_used: Final = self.is_mid_conversation_output_config_used(messages)
|
||||
is_thinking_display_updates_used: Final = self.is_thinking_display_updates_used(thinking)
|
||||
|
||||
user_beta_set: Final = set(get_anthropic_beta_from_headers(headers))
|
||||
beta_set: Final = set(user_beta_set)
|
||||
|
|
@ -268,6 +282,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")),
|
||||
custom_llm_provider="bedrock",
|
||||
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
|
||||
is_thinking_display_updates_used=is_thinking_display_updates_used,
|
||||
)
|
||||
beta_set.update(auto_betas)
|
||||
|
||||
|
|
|
|||
|
|
@ -49,6 +49,7 @@ from litellm.types.llms.anthropic import (
|
|||
ANTHROPIC_BETA_HEADER_VALUES,
|
||||
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER,
|
||||
ANTHROPIC_TOOL_SEARCH_BETA_HEADER,
|
||||
AnthropicThinkingParam,
|
||||
)
|
||||
from litellm.types.llms.bedrock import BedrockInvokeAnthropicMessagesRequest
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
|
|
@ -348,8 +349,18 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
if not isinstance(output_config, dict):
|
||||
output_config = {}
|
||||
output_config.setdefault("effort", self._effort_from_thinking_budget(budget_tokens))
|
||||
thinking: Final = anthropic_messages_request.get("thinking")
|
||||
display: Final = thinking.get("display") if isinstance(thinking, dict) else None
|
||||
anthropic_messages_request["output_config"] = output_config
|
||||
anthropic_messages_request["thinking"] = {"type": "adaptive"}
|
||||
if display is None:
|
||||
adaptive_thinking: Final[AnthropicThinkingParam] = {"type": "adaptive"}
|
||||
anthropic_messages_request["thinking"] = adaptive_thinking
|
||||
else:
|
||||
adaptive_thinking_with_display: Final[AnthropicThinkingParam] = {
|
||||
"type": "adaptive",
|
||||
"display": display,
|
||||
}
|
||||
anthropic_messages_request["thinking"] = adaptive_thinking_with_display
|
||||
verbose_logger.debug(
|
||||
"Bedrock clear_thinking_20251015: injected adaptive thinking with effort=%s for model=%s",
|
||||
output_config.get("effort"),
|
||||
|
|
@ -535,6 +546,9 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
),
|
||||
custom_llm_provider="bedrock",
|
||||
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
|
||||
is_thinking_display_updates_used=anthropic_model_info.is_thinking_display_updates_used(
|
||||
anthropic_messages_request.get("thinking")
|
||||
),
|
||||
)
|
||||
beta_set.update(auto_betas)
|
||||
|
||||
|
|
@ -664,6 +678,8 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
litellm_params: GenericLiteLLMParams,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
requested_thinking: Final = anthropic_messages_optional_request_params.get("thinking")
|
||||
requested_display_updates: Final = AnthropicModelInfo().is_thinking_display_updates_used(requested_thinking)
|
||||
self._clamp_adaptive_reasoning_effort_for_bedrock(
|
||||
model=model,
|
||||
optional_params=anthropic_messages_optional_request_params,
|
||||
|
|
@ -676,6 +692,14 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
translated_thinking: Final = anthropic_messages_request.get("thinking")
|
||||
if (
|
||||
requested_display_updates
|
||||
and isinstance(translated_thinking, dict)
|
||||
and translated_thinking.get("type") == "adaptive"
|
||||
):
|
||||
thinking_with_display: Final[AnthropicThinkingParam] = {"type": "adaptive", "display": "updates"}
|
||||
anthropic_messages_request["thinking"] = thinking_with_display
|
||||
self._normalize_system_role_messages(anthropic_messages_request, model=model)
|
||||
#########################################################
|
||||
############## BEDROCK Invoke SPECIFIC TRANSFORMATION ###
|
||||
|
|
|
|||
|
|
@ -733,7 +733,7 @@ ANTHROPIC_API_ONLY_HEADERS: Final = { # fails if calling anthropic on vertex ai
|
|||
class AnthropicThinkingParam(TypedDict, total=False):
|
||||
type: ReadOnly[Literal["enabled", "adaptive", "disabled"]]
|
||||
budget_tokens: int
|
||||
display: ReadOnly[Literal["summarized", "omitted"]]
|
||||
display: ReadOnly[Literal["summarized", "omitted", "updates"]]
|
||||
|
||||
|
||||
class ANTHROPIC_HOSTED_TOOLS(str, Enum):
|
||||
|
|
@ -776,6 +776,8 @@ ANTHROPIC_EFFORT_BETA_HEADER: Final = "effort-2025-11-24"
|
|||
|
||||
ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER: Final = "mid-conversation-output-config-2026-07-01"
|
||||
|
||||
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER: Final = "thinking-display-updates-2026-08-18"
|
||||
|
||||
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER: Final = "fine-grained-tool-streaming-2025-05-14"
|
||||
|
||||
# OAuth constants
|
||||
|
|
|
|||
|
|
@ -130,3 +130,23 @@ def test_json_provider_passthrough_adds_per_turn_control_beta():
|
|||
)
|
||||
|
||||
assert PER_TURN_CONTROL in _betas(headers)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
|
||||
@pytest.mark.parametrize("explicit_beta", (False, True))
|
||||
def test_native_messages_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
|
||||
from typing import Final
|
||||
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
|
||||
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
headers, _ = AnthropicMessagesConfig().validate_anthropic_messages_environment(
|
||||
headers={"anthropic-beta": beta} if explicit_beta else {},
|
||||
model="claude-opus-5",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
optional_params={"thinking": {"type": "adaptive", "display": display}} if display else {},
|
||||
litellm_params={},
|
||||
api_key="sk-ant-test",
|
||||
)
|
||||
|
||||
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(display == "updates" or explicit_beta)
|
||||
|
|
|
|||
|
|
@ -1984,7 +1984,6 @@ class TestClaudeOpus48AdaptiveThinking:
|
|||
|
||||
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
|
|
@ -2376,3 +2375,78 @@ def test_validate_environment_adds_mid_conversation_output_config_beta(
|
|||
|
||||
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(nested_output_config or explicit_beta)
|
||||
assert headers["x-api-key"] == FAKE_REGULAR_KEY
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
|
||||
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
|
||||
@pytest.mark.parametrize("explicit_beta", (False, True))
|
||||
def test_validate_environment_adds_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
|
||||
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
headers: Final = AnthropicModelInfo().validate_environment(
|
||||
headers={"anthropic-beta": beta} if explicit_beta else {},
|
||||
model="claude-opus-5",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
optional_params={"thinking": {"type": "adaptive", "display": display}} if display else {},
|
||||
litellm_params={},
|
||||
api_key=FAKE_REGULAR_KEY,
|
||||
)
|
||||
|
||||
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(display == "updates" or explicit_beta)
|
||||
assert headers["x-api-key"] == FAKE_REGULAR_KEY
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("thinking", "expected"),
|
||||
(
|
||||
(None, False),
|
||||
({}, False),
|
||||
("updates", False),
|
||||
({"display": "updates"}, False),
|
||||
({"type": "disabled", "display": "updates"}, False),
|
||||
({"type": "enabled", "display": "updates", "budget_tokens": 1024}, True),
|
||||
),
|
||||
)
|
||||
def test_thinking_display_beta_requires_active_thinking(thinking: object, expected: bool) -> None:
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
|
||||
headers: Final = AnthropicModelInfo().validate_environment(
|
||||
headers={},
|
||||
model="claude-opus-5",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
optional_params={"thinking": thinking},
|
||||
litellm_params={},
|
||||
api_key=FAKE_REGULAR_KEY,
|
||||
)
|
||||
|
||||
assert (ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in headers.get("anthropic-beta", "").split(",")) is expected
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize(
|
||||
("display", "expected_thinking"),
|
||||
(
|
||||
("summarized", {"type": "adaptive", "display": "summarized"}),
|
||||
("omitted", {"type": "adaptive", "display": "omitted"}),
|
||||
("updates", {"type": "adaptive"}),
|
||||
),
|
||||
)
|
||||
def test_shared_legacy_thinking_translation_preserves_supported_display(
|
||||
display: str, expected_thinking: dict[str, str]
|
||||
) -> None:
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
optional_params: Final = {
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": display},
|
||||
}
|
||||
|
||||
AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model(
|
||||
model="claude-opus-5",
|
||||
optional_params=optional_params,
|
||||
custom_llm_provider="azure_ai",
|
||||
)
|
||||
|
||||
assert optional_params["thinking"] == expected_thinking
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
import asyncio
|
||||
import base64
|
||||
import copy
|
||||
import json
|
||||
|
|
@ -12,14 +11,12 @@ import pytest
|
|||
|
||||
# Ensure the project root is on the import path so `litellm` can be imported when
|
||||
# tests are executed from any working directory.
|
||||
|
||||
import litellm
|
||||
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
|
||||
AmazonAnthropicClaudeConfig,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
|
||||
|
||||
ONE_PIXEL_PNG = base64.b64decode(
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
|
||||
)
|
||||
|
|
@ -93,9 +90,7 @@ def local_beta_headers_config(monkeypatch):
|
|||
|
||||
def test_get_supported_params_thinking():
|
||||
config = AmazonAnthropicClaudeConfig()
|
||||
params = config.get_supported_openai_params(
|
||||
model="anthropic.claude-sonnet-4-20250514-v1:0"
|
||||
)
|
||||
params = config.get_supported_openai_params(model="anthropic.claude-sonnet-4-20250514-v1:0")
|
||||
assert "thinking" in params
|
||||
|
||||
|
||||
|
|
@ -148,53 +143,23 @@ def test_aws_params_filtered_from_request_body():
|
|||
result_json = json.dumps(result)
|
||||
|
||||
# Verify AWS authentication params are NOT in the request body
|
||||
assert (
|
||||
"aws_access_key_id" not in result_json
|
||||
), "AWS access key should not be in request body"
|
||||
assert (
|
||||
"aws_secret_access_key" not in result_json
|
||||
), "AWS secret key should not be in request body"
|
||||
assert (
|
||||
"aws_session_token" not in result_json
|
||||
), "AWS session token should not be in request body"
|
||||
assert (
|
||||
"aws_region_name" not in result_json
|
||||
), "AWS region should not be in request body"
|
||||
assert (
|
||||
"aws_role_name" not in result_json
|
||||
), "AWS role name should not be in request body"
|
||||
assert (
|
||||
"aws_session_name" not in result_json
|
||||
), "AWS session name should not be in request body"
|
||||
assert (
|
||||
"aws_profile_name" not in result_json
|
||||
), "AWS profile name should not be in request body"
|
||||
assert (
|
||||
"aws_web_identity_token" not in result_json
|
||||
), "AWS web identity token should not be in request body"
|
||||
assert (
|
||||
"aws_sts_endpoint" not in result_json
|
||||
), "AWS STS endpoint should not be in request body"
|
||||
assert (
|
||||
"aws_bedrock_runtime_endpoint" not in result_json
|
||||
), "AWS bedrock endpoint should not be in request body"
|
||||
assert (
|
||||
"aws_external_id" not in result_json
|
||||
), "AWS external ID should not be in request body"
|
||||
assert (
|
||||
"aws_session_tags" not in result_json
|
||||
), "AWS session tags should not be in request body"
|
||||
assert "aws_access_key_id" not in result_json, "AWS access key should not be in request body"
|
||||
assert "aws_secret_access_key" not in result_json, "AWS secret key should not be in request body"
|
||||
assert "aws_session_token" not in result_json, "AWS session token should not be in request body"
|
||||
assert "aws_region_name" not in result_json, "AWS region should not be in request body"
|
||||
assert "aws_role_name" not in result_json, "AWS role name should not be in request body"
|
||||
assert "aws_session_name" not in result_json, "AWS session name should not be in request body"
|
||||
assert "aws_profile_name" not in result_json, "AWS profile name should not be in request body"
|
||||
assert "aws_web_identity_token" not in result_json, "AWS web identity token should not be in request body"
|
||||
assert "aws_sts_endpoint" not in result_json, "AWS STS endpoint should not be in request body"
|
||||
assert "aws_bedrock_runtime_endpoint" not in result_json, "AWS bedrock endpoint should not be in request body"
|
||||
assert "aws_external_id" not in result_json, "AWS external ID should not be in request body"
|
||||
assert "aws_session_tags" not in result_json, "AWS session tags should not be in request body"
|
||||
|
||||
# Also check that the sensitive values themselves are not in the response
|
||||
assert (
|
||||
"AKIAIOSFODNN7EXAMPLE" not in result_json
|
||||
), "AWS access key value leaked in request body"
|
||||
assert (
|
||||
"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY" not in result_json
|
||||
), "AWS secret key value leaked in request body"
|
||||
assert (
|
||||
"arn:aws:iam::123456789012:role/test-role" not in result_json
|
||||
), "AWS role ARN leaked in request body"
|
||||
assert "AKIAIOSFODNN7EXAMPLE" not in result_json, "AWS access key value leaked in request body"
|
||||
assert "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY" not in result_json, "AWS secret key value leaked in request body"
|
||||
assert "arn:aws:iam::123456789012:role/test-role" not in result_json, "AWS role ARN leaked in request body"
|
||||
assert "test-session" not in result_json, "AWS session name leaked in request body"
|
||||
|
||||
# Verify normal params ARE still in the request body
|
||||
|
|
@ -203,9 +168,7 @@ def test_aws_params_filtered_from_request_body():
|
|||
assert result["top_p"] == 0.9, "top_p should be in request body"
|
||||
|
||||
# Verify Bedrock-specific params are added
|
||||
assert (
|
||||
result["anthropic_version"] == "bedrock-2023-05-31"
|
||||
), "anthropic_version should be set"
|
||||
assert result["anthropic_version"] == "bedrock-2023-05-31", "anthropic_version should be set"
|
||||
assert "model" not in result, "model should be removed for Bedrock Invoke API"
|
||||
assert "stream" not in result, "stream should be removed for Bedrock Invoke API"
|
||||
|
||||
|
|
@ -262,9 +225,7 @@ def test_output_format_conversion_to_inline_schema():
|
|||
)
|
||||
|
||||
# Verify output_format was removed from the request
|
||||
assert (
|
||||
"output_format" not in result
|
||||
), "output_format should be removed from request body"
|
||||
assert "output_format" not in result, "output_format should be removed from request body"
|
||||
|
||||
# Verify the schema was added to the last user message content
|
||||
assert "messages" in result
|
||||
|
|
@ -415,9 +376,7 @@ def test_opus_4_5_model_detection():
|
|||
]
|
||||
|
||||
for model in non_opus_4_5_models:
|
||||
assert not config._is_claude_opus_4_5(
|
||||
model
|
||||
), f"Should not detect {model} as Opus 4.5"
|
||||
assert not config._is_claude_opus_4_5(model), f"Should not detect {model} as Opus 4.5"
|
||||
|
||||
|
||||
# def test_structured_outputs_beta_header_filtered_for_bedrock_invoke():
|
||||
|
|
@ -595,9 +554,7 @@ def test_output_config_format_forwarded_for_bedrock_chat_invoke_request(local_mo
|
|||
("anthropic.claude-opus-4-7", "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_output_config_effort_normalized_for_bedrock_chat_invoke_request(
|
||||
model, expected_effort
|
||||
):
|
||||
def test_output_config_effort_normalized_for_bedrock_chat_invoke_request(model, expected_effort):
|
||||
"""Bedrock Invoke chat path accepts ``xhigh`` and forwards the provider-safe effort."""
|
||||
config = AmazonAnthropicClaudeConfig()
|
||||
|
||||
|
|
@ -668,9 +625,9 @@ def test_output_format_removed_from_bedrock_invoke_request():
|
|||
)
|
||||
|
||||
# Verify output_format is not in the request
|
||||
assert (
|
||||
"output_format" not in result
|
||||
), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
|
||||
assert "output_format" not in result, (
|
||||
f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
|
||||
)
|
||||
|
||||
|
||||
def test_bedrock_chat_invoke_forwards_output_config_format_natively(local_model_cost_map):
|
||||
|
|
@ -866,7 +823,9 @@ async def test_bedrock_invoke_claude_async_completion_inlines_remote_images_off_
|
|||
assert async_only_image_fetch.base64_png in captured["body"]
|
||||
|
||||
|
||||
async def test_bedrock_invoke_claude_async_completion_inlines_document_url_sources_off_the_event_loop(async_only_image_fetch):
|
||||
async def test_bedrock_invoke_claude_async_completion_inlines_document_url_sources_off_the_event_loop(
|
||||
async_only_image_fetch,
|
||||
):
|
||||
pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf"
|
||||
captured = {}
|
||||
|
||||
|
|
@ -958,6 +917,62 @@ def test_bedrock_chat_invoke_tool_search_beta_follows_model_map(
|
|||
assert result.get("anthropic_beta") == expected_betas
|
||||
|
||||
|
||||
def test_bedrock_chat_invoke_adds_thinking_display_updates_beta(
|
||||
local_model_cost_map, local_beta_headers_config
|
||||
) -> None:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
|
||||
config: Final = AmazonAnthropicClaudeConfig()
|
||||
model: Final = "us.anthropic.claude-opus-5"
|
||||
optional_params: Final = config.map_openai_params(
|
||||
non_default_params={
|
||||
"max_tokens": 512,
|
||||
"thinking": {"type": "adaptive", "display": "updates"},
|
||||
},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
result: Final = config.transform_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
|
||||
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
|
||||
|
||||
|
||||
def test_bedrock_chat_invoke_preserves_display_when_translating_legacy_thinking(
|
||||
local_model_cost_map, local_beta_headers_config
|
||||
) -> None:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
|
||||
config: Final = AmazonAnthropicClaudeConfig()
|
||||
model: Final = "us.anthropic.claude-opus-5"
|
||||
optional_params: Final = config.map_openai_params(
|
||||
non_default_params={
|
||||
"max_tokens": 512,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": "updates"},
|
||||
},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
result: Final = config.transform_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
|
||||
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
|
||||
|
||||
|
||||
FINE_GRAINED_TOOL_STREAMING_BETA: Final = "fine-grained-tool-streaming-2025-05-14"
|
||||
EAGER_TOOL_SCHEMA: Final = {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}
|
||||
|
||||
|
|
@ -1015,7 +1030,10 @@ def test_bedrock_chat_invoke_eager_input_streaming_beta_not_duplicated_with_clie
|
|||
|
||||
def _mid_conversation_system_conversation() -> list[dict]:
|
||||
return [
|
||||
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
|
||||
{
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}],
|
||||
},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
|
|
@ -1074,7 +1092,11 @@ def _preserved_thinking_turns(reminder_after_user: bool) -> tuple[list[dict], li
|
|||
second_question = {"role": "user", "content": "Second question"}
|
||||
second_turn = [second_question, reminder] if reminder_after_user else [reminder, second_question]
|
||||
turn_n_plus_one = [*turn_n, _thinking_reply("First answer"), *second_turn]
|
||||
turn_n_plus_two = [*turn_n_plus_one, _thinking_reply("Second answer"), {"role": "user", "content": "Third question"}]
|
||||
turn_n_plus_two = [
|
||||
*turn_n_plus_one,
|
||||
_thinking_reply("Second answer"),
|
||||
{"role": "user", "content": "Third question"},
|
||||
]
|
||||
return turn_n, turn_n_plus_one, turn_n_plus_two
|
||||
|
||||
|
||||
|
|
@ -1102,7 +1124,11 @@ def test_chat_flagged_model_replays_a_byte_identical_prefix_around_a_mid_convers
|
|||
request must be a byte-identical prefix of turn N+1's or the block is dropped."""
|
||||
requests = [
|
||||
AmazonAnthropicClaudeConfig().transform_request(
|
||||
model="invoke/us.anthropic.claude-fable-5-1", messages=copy.deepcopy(turn), optional_params={}, litellm_params={}, headers={}
|
||||
model="invoke/us.anthropic.claude-fable-5-1",
|
||||
messages=copy.deepcopy(turn),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
for turn in _preserved_thinking_turns(reminder_after_user)
|
||||
]
|
||||
|
|
|
|||
|
|
@ -5,33 +5,33 @@ import json
|
|||
import os
|
||||
import struct
|
||||
import zlib
|
||||
from collections.abc import AsyncIterator, Mapping, Sequence
|
||||
from datetime import datetime
|
||||
from types import SimpleNamespace
|
||||
from collections.abc import AsyncIterator, Mapping, Sequence
|
||||
from typing import Final
|
||||
from unittest.mock import Mock
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
# Ensure the project root is on the import path so `litellm` can be imported when
|
||||
# tests are executed from any working directory.
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.bedrock.common_utils import (
|
||||
ensure_bedrock_anthropic_messages_tool_names,
|
||||
normalize_custom_field_on_tools,
|
||||
normalize_tool_input_schema_types_for_bedrock_invoke,
|
||||
)
|
||||
from litellm.constants import (
|
||||
BEDROCK_MIN_THINKING_BUDGET_TOKENS,
|
||||
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
|
||||
)
|
||||
|
||||
# Ensure the project root is on the import path so `litellm` can be imported when
|
||||
# tests are executed from any working directory.
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.anthropic.pass_through.messages.mid_conversation_system import (
|
||||
as_system_content_blocks,
|
||||
)
|
||||
from litellm.llms.bedrock.common_utils import (
|
||||
ensure_bedrock_anthropic_messages_tool_names,
|
||||
normalize_custom_field_on_tools,
|
||||
normalize_tool_input_schema_types_for_bedrock_invoke,
|
||||
)
|
||||
from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_transformation import (
|
||||
AmazonAnthropicClaudeMessagesConfig,
|
||||
AmazonAnthropicClaudeMessagesStreamDecoder,
|
||||
|
|
@ -54,9 +54,7 @@ async def test_bedrock_sse_wrapper_encodes_dict_chunks():
|
|||
_dummy_stream(),
|
||||
litellm_logging_obj=LiteLLMLoggingObj(
|
||||
model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hello, can you tell me a short joke?"}
|
||||
],
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
stream=True,
|
||||
call_type="chat",
|
||||
start_time=datetime.now(),
|
||||
|
|
@ -233,9 +231,7 @@ async def test_bedrock_sse_wrapper_keeps_usage_in_message_start_and_message_delt
|
|||
def test_chunk_parser_usage_transformation():
|
||||
"""Ensure Bedrock invocation metrics are transformed to Anthropic usage keys."""
|
||||
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
|
||||
model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0"
|
||||
)
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0")
|
||||
|
||||
chunk = {
|
||||
"type": "message_delta",
|
||||
|
|
@ -264,9 +260,7 @@ def test_chunk_parser_preserves_cache_usage_fields_with_invocation_metrics():
|
|||
fields and cache tokens end up billed at $0.
|
||||
"""
|
||||
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
|
||||
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
|
||||
)
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
|
||||
|
||||
chunk = {
|
||||
"type": "message_stop",
|
||||
|
|
@ -292,9 +286,7 @@ def test_chunk_parser_preserves_cache_usage_fields_with_invocation_metrics():
|
|||
def test_chunk_parser_maps_cache_token_counts_from_invocation_metrics():
|
||||
"""Cache itemization inside invocationMetrics maps to Anthropic usage keys."""
|
||||
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
|
||||
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
|
||||
)
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
|
||||
|
||||
chunk = {
|
||||
"type": "message_stop",
|
||||
|
|
@ -317,9 +309,7 @@ def test_chunk_parser_maps_cache_token_counts_from_invocation_metrics():
|
|||
def test_chunk_parser_keeps_existing_token_counts_over_invocation_metrics():
|
||||
"""Token counts reported in the chunk's own usage block win over invocationMetrics."""
|
||||
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
|
||||
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
|
||||
)
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
|
||||
|
||||
chunk = {
|
||||
"type": "message_stop",
|
||||
|
|
@ -354,9 +344,7 @@ async def test_bedrock_sse_wrapper_preserves_cache_usage_with_invocation_metrics
|
|||
final usage billed cache reads and writes at $0.
|
||||
"""
|
||||
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
|
||||
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
|
||||
)
|
||||
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
||||
raw_chunks = [
|
||||
|
|
@ -566,11 +554,7 @@ def test_normalize_custom_field_on_tools():
|
|||
assert request4["tools"] is None
|
||||
|
||||
# Case 5: an explicit top-level flag wins over a conflicting wrapped one
|
||||
request5 = {
|
||||
"tools": [
|
||||
{"name": "Read", "defer_loading": False, "custom": {"defer_loading": True}}
|
||||
]
|
||||
}
|
||||
request5 = {"tools": [{"name": "Read", "defer_loading": False, "custom": {"defer_loading": True}}]}
|
||||
normalize_custom_field_on_tools(request5)
|
||||
assert request5["tools"][0] == {"name": "Read", "defer_loading": False}
|
||||
|
||||
|
|
@ -591,9 +575,7 @@ def test_normalize_custom_field_on_tools():
|
|||
assert request7["tools"] == [{"name": "Read"}, {"name": "Write"}]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"deferred_marker", [{"custom": {"defer_loading": True}}, {"defer_loading": True}]
|
||||
)
|
||||
@pytest.mark.parametrize("deferred_marker", [{"custom": {"defer_loading": True}}, {"defer_loading": True}])
|
||||
def test_bedrock_invoke_messages_transform_emits_top_level_defer_loading(
|
||||
deferred_marker,
|
||||
):
|
||||
|
|
@ -726,9 +708,7 @@ def test_bedrock_invoke_messages_skips_thinking_injection_when_already_enabled(
|
|||
"max_tokens": 32000,
|
||||
"stream": False,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048},
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
model="global.anthropic.claude-sonnet-4-6-v1:0",
|
||||
|
|
@ -830,9 +810,7 @@ def test_remove_ttl_from_cache_control_processes_tools(local_model_cost_map):
|
|||
"messages": [],
|
||||
}
|
||||
|
||||
cfg._remove_ttl_from_cache_control(
|
||||
request, model="anthropic.claude-3-5-sonnet-20241022-v2:0"
|
||||
)
|
||||
cfg._remove_ttl_from_cache_control(request, model="anthropic.claude-3-5-sonnet-20241022-v2:0")
|
||||
|
||||
# Tool ttl should be stripped
|
||||
assert "ttl" not in request["tools"][0]["cache_control"]
|
||||
|
|
@ -868,9 +846,7 @@ def test_remove_ttl_from_cache_control_preserves_tools_ttl_for_claude_4_5(local_
|
|||
],
|
||||
}
|
||||
|
||||
cfg._remove_ttl_from_cache_control(
|
||||
request, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||
)
|
||||
cfg._remove_ttl_from_cache_control(request, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0")
|
||||
|
||||
# Both tools and system should preserve ttl for Claude 4.5
|
||||
assert request["tools"][0]["cache_control"]["ttl"] == "1h"
|
||||
|
|
@ -954,9 +930,7 @@ def test_bedrock_messages_strips_output_config():
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert "output_config" not in result, (
|
||||
"output_config should be stripped for models that don't support it"
|
||||
)
|
||||
assert "output_config" not in result, "output_config should be stripped for models that don't support it"
|
||||
assert result.get("max_tokens") == 4096
|
||||
|
||||
|
||||
|
|
@ -989,9 +963,7 @@ def test_bedrock_messages_preserves_output_config_for_claude_4_6():
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert "output_config" in result, (
|
||||
"output_config should be preserved for supported models"
|
||||
)
|
||||
assert "output_config" in result, "output_config should be preserved for supported models"
|
||||
assert result["output_config"] == {"effort": "high"}
|
||||
assert result.get("max_tokens") == 4096
|
||||
|
||||
|
|
@ -1143,9 +1115,7 @@ def test_bedrock_messages_converts_output_config_format_to_inline_schema():
|
|||
("anthropic.claude-opus-4-7", "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_bedrock_messages_normalizes_output_config_effort_for_opus(
|
||||
model, expected_effort
|
||||
):
|
||||
def test_bedrock_messages_normalizes_output_config_effort_for_opus(model, expected_effort):
|
||||
"""Bedrock /v1/messages accepts ``xhigh`` and forwards the provider-safe effort."""
|
||||
from unittest.mock import patch
|
||||
|
||||
|
|
@ -1203,9 +1173,7 @@ def test_bedrock_messages_does_not_mutate_callers_messages_when_embedding_schema
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert caller_messages == [
|
||||
{"role": "user", "content": [{"type": "text", "text": "Hello"}]}
|
||||
]
|
||||
assert caller_messages == [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
|
||||
assert caller_message == {
|
||||
"role": "user",
|
||||
"content": [{"type": "text", "text": "Hello"}],
|
||||
|
|
@ -1521,9 +1489,7 @@ def test_bedrock_messages_strips_context_management():
|
|||
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
|
|
@ -1534,9 +1500,7 @@ def test_bedrock_messages_strips_context_management():
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert "context_management" not in result, (
|
||||
"context_management should be stripped — Bedrock Invoke rejects it"
|
||||
)
|
||||
assert "context_management" not in result, "context_management should be stripped — Bedrock Invoke rejects it"
|
||||
assert result.get("max_tokens") == 4096
|
||||
|
||||
|
||||
|
|
@ -1661,7 +1625,9 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
|
|||
["dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14", "interleaved-thinking-2025-05-14"],
|
||||
ids=["client_sends_beta", "client_omits_beta"],
|
||||
)
|
||||
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(local_beta_headers_config, client_beta_header):
|
||||
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(
|
||||
local_beta_headers_config, client_beta_header
|
||||
):
|
||||
"""
|
||||
Claude Code's server-side auto-mode classifier sends `safeguards` alongside the
|
||||
dangerous-tool-use-2026-09-03 beta. Bedrock Invoke accepts the pair, answers
|
||||
|
|
@ -1769,12 +1735,8 @@ def test_bedrock_messages_filters_user_provided_unsupported_beta_header():
|
|||
)
|
||||
|
||||
betas = result.get("anthropic_beta") or []
|
||||
assert "advisor-tool-2026-03-01" not in betas, (
|
||||
"user-provided beta not in the Bedrock mapping must be dropped"
|
||||
)
|
||||
assert "context-1m-2025-08-07" in betas, (
|
||||
"user-provided beta that IS in the Bedrock mapping should survive"
|
||||
)
|
||||
assert "advisor-tool-2026-03-01" not in betas, "user-provided beta not in the Bedrock mapping must be dropped"
|
||||
assert "context-1m-2025-08-07" in betas, "user-provided beta that IS in the Bedrock mapping should survive"
|
||||
|
||||
|
||||
def test_bedrock_messages_renames_user_provided_aliased_beta_header():
|
||||
|
|
@ -1802,9 +1764,7 @@ def test_bedrock_messages_renames_user_provided_aliased_beta_header():
|
|||
assert "advanced-tool-use-2025-11-20" not in betas, (
|
||||
"Anthropic-direct spelling should be rewritten, not forwarded verbatim"
|
||||
)
|
||||
assert "tool-search-tool-2025-10-19" in betas, (
|
||||
"user-provided beta should be renamed to the Bedrock-side spelling"
|
||||
)
|
||||
assert "tool-search-tool-2025-10-19" in betas, "user-provided beta should be renamed to the Bedrock-side spelling"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -2066,9 +2026,7 @@ async def test_unified_bedrock_messages_sse_usage_and_cost_claude_sonnet_46():
|
|||
"global.anthropic.claude-fable-5",
|
||||
],
|
||||
)
|
||||
def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models(
|
||||
local_model_cost_map, model
|
||||
):
|
||||
def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models(local_model_cost_map, model):
|
||||
"""clear_thinking_20251015 without a top-level ``thinking`` field must inject
|
||||
``thinking.type=adaptive`` plus ``output_config.effort`` on adaptive-thinking
|
||||
models (Opus 4.7/4.8, Fable 5). The legacy ``thinking.type=enabled`` shape is
|
||||
|
|
@ -2078,9 +2036,7 @@ def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models
|
|||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
request = {
|
||||
"max_tokens": 32000,
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2103,9 +2059,7 @@ def test_bedrock_clear_thinking_converts_legacy_enabled_budget_to_effort():
|
|||
"type": "enabled",
|
||||
"budget_tokens": DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
},
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2123,10 +2077,7 @@ def test_resolve_clear_thinking_budget_tokens_honors_explicit_zero():
|
|||
and only fall back to the minimum when the caller omits the budget."""
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
assert cfg._resolve_clear_thinking_budget_tokens(0) == 0
|
||||
assert (
|
||||
cfg._resolve_clear_thinking_budget_tokens(None)
|
||||
== BEDROCK_MIN_THINKING_BUDGET_TOKENS
|
||||
)
|
||||
assert cfg._resolve_clear_thinking_budget_tokens(None) == BEDROCK_MIN_THINKING_BUDGET_TOKENS
|
||||
assert cfg._resolve_clear_thinking_budget_tokens(12000) == 12000
|
||||
|
||||
|
||||
|
|
@ -2136,9 +2087,7 @@ def test_bedrock_clear_thinking_keeps_enabled_for_non_adaptive_models():
|
|||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
request = {
|
||||
"max_tokens": 32000,
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2163,9 +2112,7 @@ def test_bedrock_invoke_transform_emits_adaptive_thinking_for_opus_4_8():
|
|||
optional_params = {
|
||||
"max_tokens": 32000,
|
||||
"stream": False,
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
|
|
@ -2202,9 +2149,7 @@ def test_bedrock_invoke_transform_normalizes_system_role_message_into_system():
|
|||
|
||||
assert all(m.get("role") != "system" for m in result["messages"])
|
||||
assert result["messages"] == [{"role": "user", "content": "hi"}]
|
||||
assert result["system"] == [
|
||||
{"type": "text", "text": "You are a careful assistant."}
|
||||
]
|
||||
assert result["system"] == [{"type": "text", "text": "You are a careful assistant."}]
|
||||
|
||||
|
||||
def test_bedrock_invoke_transform_merges_system_role_into_existing_system():
|
||||
|
|
@ -2319,9 +2264,7 @@ def test_bedrock_invoke_transform_keeps_mid_conversation_system_role_in_place(lo
|
|||
)
|
||||
|
||||
assert result["messages"] == messages
|
||||
assert result["system"] == [
|
||||
{"type": "text", "text": "Base.", "cache_control": {"type": "ephemeral"}}
|
||||
]
|
||||
assert result["system"] == [{"type": "text", "text": "Base.", "cache_control": {"type": "ephemeral"}}]
|
||||
|
||||
|
||||
def test_bedrock_invoke_transform_hoists_only_leading_system_run(local_model_cost_map):
|
||||
|
|
@ -2504,13 +2447,13 @@ def test_bedrock_invoke_transform_converted_system_carries_only_its_content(loca
|
|||
assert result["messages"][2] == {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": (
|
||||
"Operator note (not from the user): the following was "
|
||||
"originally a mid-conversation system-role reminder."
|
||||
),
|
||||
},
|
||||
{
|
||||
"type": "text",
|
||||
"text": (
|
||||
"Operator note (not from the user): the following was "
|
||||
"originally a mid-conversation system-role reminder."
|
||||
),
|
||||
},
|
||||
{"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"},
|
||||
],
|
||||
}
|
||||
|
|
@ -2646,10 +2589,7 @@ def test_as_system_content_blocks_handles_each_shape():
|
|||
def test_effort_from_thinking_budget_tiers(budget_tokens, expected_effort):
|
||||
"""The budget -> effort mapping pins each tier boundary so a shifted threshold
|
||||
is caught."""
|
||||
assert (
|
||||
AmazonAnthropicClaudeMessagesConfig._effort_from_thinking_budget(budget_tokens)
|
||||
== expected_effort
|
||||
)
|
||||
assert AmazonAnthropicClaudeMessagesConfig._effort_from_thinking_budget(budget_tokens) == expected_effort
|
||||
|
||||
|
||||
def test_inject_adaptive_thinking_preserves_existing_effort():
|
||||
|
|
@ -2658,9 +2598,7 @@ def test_inject_adaptive_thinking_preserves_existing_effort():
|
|||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
request = {"output_config": {"effort": "max", "other": "keep"}}
|
||||
|
||||
cfg._inject_adaptive_thinking_for_clear_thinking(
|
||||
request, budget_tokens=24000, model="us.anthropic.claude-fable-5"
|
||||
)
|
||||
cfg._inject_adaptive_thinking_for_clear_thinking(request, budget_tokens=24000, model="us.anthropic.claude-fable-5")
|
||||
|
||||
assert request["thinking"] == {"type": "adaptive"}
|
||||
assert request["output_config"] == {"effort": "max", "other": "keep"}
|
||||
|
|
@ -2673,9 +2611,7 @@ def test_bedrock_clear_thinking_noops_when_thinking_already_adaptive():
|
|||
request = {
|
||||
"max_tokens": 32000,
|
||||
"thinking": {"type": "adaptive"},
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2695,9 +2631,7 @@ def test_bedrock_clear_thinking_replaces_disabled_thinking_on_adaptive_model():
|
|||
request = {
|
||||
"max_tokens": 32000,
|
||||
"thinking": {"type": "disabled"},
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2717,9 +2651,7 @@ def test_bedrock_clear_thinking_leaves_enabled_thinking_on_non_adaptive_model():
|
|||
request = {
|
||||
"max_tokens": 32000,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 8000},
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
|
||||
}
|
||||
|
||||
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
|
||||
|
|
@ -2754,9 +2686,7 @@ def test_bedrock_messages_preserves_clear_tool_uses_context_management_and_adds_
|
|||
messages = [{"role": "user", "content": [{"type": "text", "text": "Hi"}]}]
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"context_management": {
|
||||
"edits": [{"type": "clear_tool_uses_20250919"}]
|
||||
},
|
||||
"context_management": {"edits": [{"type": "clear_tool_uses_20250919"}]},
|
||||
}
|
||||
|
||||
result = cfg.transform_anthropic_messages_request(
|
||||
|
|
@ -2767,12 +2697,11 @@ def test_bedrock_messages_preserves_clear_tool_uses_context_management_and_adds_
|
|||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("context_management") == {
|
||||
"edits": [{"type": "clear_tool_uses_20250919"}]
|
||||
}, "clear_tool_uses_20250919 edit must reach Bedrock InvokeModel body"
|
||||
assert result.get("context_management") == {"edits": [{"type": "clear_tool_uses_20250919"}]}, (
|
||||
"clear_tool_uses_20250919 edit must reach Bedrock InvokeModel body"
|
||||
)
|
||||
assert "context-management-2025-06-27" in result.get("anthropic_beta", []), (
|
||||
"context-management-2025-06-27 beta must reach the InvokeModel body so "
|
||||
"the tool-call-clearing edit is accepted"
|
||||
"context-management-2025-06-27 beta must reach the InvokeModel body so the tool-call-clearing edit is accepted"
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -2849,9 +2778,9 @@ def test_bedrock_messages_filters_clear_thinking_keeps_clear_tool_uses(
|
|||
|
||||
cm = result.get("context_management")
|
||||
assert cm is not None
|
||||
assert [e.get("type") for e in cm["edits"]] == [
|
||||
"clear_tool_uses_20250919"
|
||||
], "clear_thinking_20251015 must still be stripped (LiteLLM-internal)"
|
||||
assert [e.get("type") for e in cm["edits"]] == ["clear_tool_uses_20250919"], (
|
||||
"clear_thinking_20251015 must still be stripped (LiteLLM-internal)"
|
||||
)
|
||||
|
||||
betas = result.get("anthropic_beta", [])
|
||||
assert "context-management-2025-06-27" in betas
|
||||
|
|
@ -2992,9 +2921,7 @@ def test_bedrock_messages_tool_search_follows_claude_tool_search_rule(local_mode
|
|||
assert cfg._supports_tool_search_on_bedrock(model) is expected
|
||||
|
||||
|
||||
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
|
||||
local_model_cost_map, monkeypatch
|
||||
):
|
||||
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(local_model_cost_map, monkeypatch):
|
||||
"""The outbound thinking payload must follow the exact Bedrock cost-map entry.
|
||||
Before threading the caller's provider through the capability probes, the probe
|
||||
was pinned to ``"anthropic"``: the exact ``global.anthropic.claude-opus-4-8``
|
||||
|
|
@ -3002,7 +2929,6 @@ def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
|
|||
forced ``thinking.type='adaptive'`` even with ``supports_adaptive_thinking``
|
||||
explicitly set to ``false`` on the entry."""
|
||||
import litellm
|
||||
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
model = "global.anthropic.claude-opus-4-8"
|
||||
|
|
@ -3404,22 +3330,14 @@ def test_bedrock_invoke_eager_input_streaming_beta_not_duplicated_with_client_he
|
|||
|
||||
def _bedrock_event_frame(payload: Mapping[str, object]) -> bytes:
|
||||
def _header(name: str, value: str) -> bytes:
|
||||
return (
|
||||
bytes([len(name)])
|
||||
+ name.encode()
|
||||
+ bytes([7])
|
||||
+ struct.pack(">H", len(value))
|
||||
+ value.encode()
|
||||
)
|
||||
return bytes([len(name)]) + name.encode() + bytes([7]) + struct.pack(">H", len(value)) + value.encode()
|
||||
|
||||
headers: Final = (
|
||||
_header(":message-type", "event")
|
||||
+ _header(":event-type", "chunk")
|
||||
+ _header(":content-type", "application/json")
|
||||
)
|
||||
body: Final = json.dumps(
|
||||
{"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}
|
||||
).encode()
|
||||
body: Final = json.dumps({"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}).encode()
|
||||
prelude: Final = struct.pack(">II", 12 + len(headers) + len(body) + 4, len(headers))
|
||||
prelude_crc: Final = struct.pack(">I", zlib.crc32(prelude))
|
||||
message_crc: Final = struct.pack(">I", zlib.crc32(prelude + prelude_crc + headers + body))
|
||||
|
|
@ -3547,3 +3465,65 @@ def test_bedrock_messages_removed_output_config_does_not_add_beta(explicit_beta:
|
|||
|
||||
assert result["messages"] == [{"role": "user", "content": "Reply with OK"}]
|
||||
assert result.get("anthropic_beta", []).count(beta) == int(explicit_beta)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
|
||||
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
|
||||
@pytest.mark.parametrize("explicit_beta", (False, True))
|
||||
def test_bedrock_messages_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
thinking: Final = {"type": "adaptive", "display": display} if display else None
|
||||
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
|
||||
model="eu.anthropic.claude-opus-5",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
anthropic_messages_optional_request_params={"max_tokens": 512, **({"thinking": thinking} if thinking else {})},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"anthropic-beta": beta} if explicit_beta else {},
|
||||
)
|
||||
|
||||
assert result.get("anthropic_beta", []).count(beta) == int(display == "updates" or explicit_beta)
|
||||
assert result.get("thinking") == thinking
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
|
||||
def test_bedrock_messages_preserves_display_when_translating_legacy_thinking() -> None:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
|
||||
model="eu.anthropic.claude-opus-5",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
anthropic_messages_optional_request_params={
|
||||
"max_tokens": 512,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 24000, "display": "updates"},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
|
||||
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
|
||||
def test_bedrock_clear_thinking_preserves_display_updates() -> None:
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
|
||||
model="us.anthropic.claude-opus-4-6",
|
||||
messages=[{"role": "user", "content": "Reply with OK"}],
|
||||
anthropic_messages_optional_request_params={
|
||||
"max_tokens": 512,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": "updates"},
|
||||
"context_management": {"edits": [{"type": "clear_thinking_20251015"}]},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
|
||||
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue