fix(bedrock): add beta for thinking display updates (#43832)

This commit is contained in:
shrey-berri 2026-10-01 09:21:10 -07:00 • committed by GitHub
parent 4b06d04334
commit c2ae483782
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 395 additions and 225 deletions

View file

@ -34,9 +34,11 @@ from litellm.types.llms.anthropic import (
ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER,
ANTHROPIC_OAUTH_BETA_HEADER,
ANTHROPIC_OAUTH_TOKEN_PREFIX,
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,
AllAnthropicToolsValues,
AnthropicMcpServerTool,
AnthropicMessagesToolChoice,
AnthropicThinkingParam,
)
from litellm.types.llms.openai import AllMessageValues
from litellm.types.proxy.model_listing import ModelInfoResponse
@ -337,6 +339,11 @@ class AnthropicModelInfo(BaseLLMModelInfo):
file_ids: Final = get_file_ids_from_messages(messages)
return len(file_ids) > 0
def is_thinking_display_updates_used(self, thinking: AnthropicThinkingParam | None) -> bool:
if not isinstance(thinking, dict):
return False
return thinking.get("type") in ("adaptive", "enabled") and thinking.get("display") == "updates"
def is_mid_conversation_output_config_used(self, messages: list[AllMessageValues]) -> bool:
"""
Return if "output_config" is in a message
@ -749,7 +756,11 @@ class AnthropicModelInfo(BaseLLMModelInfo):
custom_llm_provider=custom_llm_provider,
)
existing_output_config: Final = optional_params.get("output_config")
optional_params["thinking"] = {"type": "adaptive"}
display: Final = thinking.get("display")
if display in ("summarized", "omitted"):
optional_params["thinking"] = {"type": "adaptive", "display": display}
else:
optional_params["thinking"] = {"type": "adaptive"}
optional_params["output_config"] = {
"effort": effort,
**(existing_output_config if isinstance(existing_output_config, dict) else MappingProxyType({})),
@ -869,6 +880,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
*,
custom_llm_provider: str,
is_mid_conversation_output_config_used: bool = False,
is_thinking_display_updates_used: bool = False,
) -> list[str]:
"""
Get list of common beta headers based on the features that are active.
@ -904,7 +916,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
if is_mid_conversation_output_config_used:
betas.append(ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER)
return list(set(betas))
thinking_display_betas: Final = (
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,) if is_thinking_display_updates_used else ()
)
return list(set(betas).union(thinking_display_betas))
@staticmethod
def _make_api_key_auth_header(api_key: str, api_base: str | None, use_bearer_for_custom_base: bool = False) -> dict:
@ -937,6 +952,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
api_base: str | None = None,
use_bearer_for_custom_base: bool = False,
is_mid_conversation_output_config_used: bool = False,
is_thinking_display_updates_used: bool = False,
) -> dict:
betas: Final = set()
# Anthropic no longer requires the prompt-caching beta header
@ -993,6 +1009,10 @@ class AnthropicModelInfo(BaseLLMModelInfo):
if user_anthropic_beta_headers is not None:
betas.update(user_anthropic_beta_headers)
all_betas: Final = betas.union(
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,) if is_thinking_display_updates_used else ()
)
# Don't send any beta headers to Vertex, except web search which is required
if is_vertex_request is True:
# Vertex AI requires web search beta header for web search to work
@ -1000,8 +1020,8 @@ class AnthropicModelInfo(BaseLLMModelInfo):
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
headers["anthropic-beta"] = ANTHROPIC_BETA_HEADER_VALUES.WEB_SEARCH_2025_03_05.value
elif len(betas) > 0:
headers["anthropic-beta"] = ",".join(betas)
elif len(all_betas) > 0:
headers["anthropic-beta"] = ",".join(all_betas)
return headers
@ -1059,6 +1079,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
auth_token=auth_token,
file_id_used=file_id_used,
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
is_thinking_display_updates_used=self.is_thinking_display_updates_used(optional_params.get("thinking")),
web_search_tool_used=web_search_tool_used,
is_vertex_request=optional_params.get("is_vertex_request", False),
user_anthropic_beta_headers=user_anthropic_beta_headers,

View file

@ -12,6 +12,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
from litellm.types.llms.anthropic import (
ANTHROPIC_ADVISOR_TOOL_TYPE,
ANTHROPIC_BETA_HEADER_VALUES,
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,
AnthropicMessagesRequest,
)
from litellm.types.llms.anthropic_messages.anthropic_response import (
@ -688,8 +689,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if AnthropicModelInfo().is_tool_search_used(tools):
beta_values.add(get_tool_search_beta_header(custom_llm_provider))
if not beta_values:
thinking_display_betas: Final = (
(ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER,)
if AnthropicModelInfo().is_thinking_display_updates_used(optional_params.get("thinking"))
else ()
)
all_beta_values: Final = beta_values.union(thinking_display_betas)
if not all_beta_values:
return headers
merged: Final = {key: value for key, value in headers.items() if key.lower() != "anthropic-beta"}
merged["anthropic-beta"] = ",".join(sorted(beta_values))
merged["anthropic-beta"] = ",".join(sorted(all_beta_values))
return merged

View file

@ -29,6 +29,7 @@ from litellm.llms.bedrock.common_utils import (
from litellm.types.llms.anthropic import (
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER,
ANTHROPIC_TOOL_SEARCH_BETA_HEADER,
AnthropicThinkingParam,
)
from litellm.types.llms.openai import AllMessageValues
from litellm.types.utils import ModelResponse
@ -85,6 +86,8 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
from litellm.utils import supports_native_structured_output
original_model: Final = model
requested_thinking: Final = non_default_params.get("thinking")
requested_display_updates: Final = self.is_thinking_display_updates_used(requested_thinking)
if "response_format" in non_default_params and not supports_native_structured_output(
model=model, custom_llm_provider="bedrock"
):
@ -114,6 +117,14 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model(
model=original_model, optional_params=optional_params, custom_llm_provider="bedrock"
)
translated_thinking: Final = optional_params.get("thinking")
if (
requested_display_updates
and isinstance(translated_thinking, dict)
and translated_thinking.get("type") == "adaptive"
):
thinking_with_display: Final[AnthropicThinkingParam] = {"type": "adaptive", "display": "updates"}
optional_params["thinking"] = thinking_with_display
# The stub model hides the original model from the parent's forced-tool-use backstop
response_format_tool_choice: Final = optional_params.get("tool_choice")
@ -170,6 +181,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
messages=messages,
optional_params=optional_params,
headers=headers,
thinking=_anthropic_request.get("thinking"),
)
if beta_list:
_anthropic_request["anthropic_beta"] = beta_list
@ -250,12 +262,14 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
messages: list[AllMessageValues],
optional_params: dict,
headers: dict,
thinking: AnthropicThinkingParam | None,
) -> list[str]:
tools: Final = optional_params.get("tools")
tool_search_used: Final = self.is_tool_search_used(tools)
programmatic_tool_calling_used: Final = self.is_programmatic_tool_calling_used(tools)
input_examples_used: Final = self.is_input_examples_used(tools)
is_mid_conversation_output_config_used: Final = self.is_mid_conversation_output_config_used(messages)
is_thinking_display_updates_used: Final = self.is_thinking_display_updates_used(thinking)
user_beta_set: Final = set(get_anthropic_beta_from_headers(headers))
beta_set: Final = set(user_beta_set)
@ -268,6 +282,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")),
custom_llm_provider="bedrock",
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
is_thinking_display_updates_used=is_thinking_display_updates_used,
)
beta_set.update(auto_betas)

View file

@ -49,6 +49,7 @@ from litellm.types.llms.anthropic import (
ANTHROPIC_BETA_HEADER_VALUES,
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER,
ANTHROPIC_TOOL_SEARCH_BETA_HEADER,
AnthropicThinkingParam,
)
from litellm.types.llms.bedrock import BedrockInvokeAnthropicMessagesRequest
from litellm.types.llms.openai import AllMessageValues
@ -348,8 +349,18 @@ class AmazonAnthropicClaudeMessagesConfig(
if not isinstance(output_config, dict):
output_config = {}
output_config.setdefault("effort", self._effort_from_thinking_budget(budget_tokens))
thinking: Final = anthropic_messages_request.get("thinking")
display: Final = thinking.get("display") if isinstance(thinking, dict) else None
anthropic_messages_request["output_config"] = output_config
anthropic_messages_request["thinking"] = {"type": "adaptive"}
if display is None:
adaptive_thinking: Final[AnthropicThinkingParam] = {"type": "adaptive"}
anthropic_messages_request["thinking"] = adaptive_thinking
else:
adaptive_thinking_with_display: Final[AnthropicThinkingParam] = {
"type": "adaptive",
"display": display,
}
anthropic_messages_request["thinking"] = adaptive_thinking_with_display
verbose_logger.debug(
"Bedrock clear_thinking_20251015: injected adaptive thinking with effort=%s for model=%s",
output_config.get("effort"),
@ -535,6 +546,9 @@ class AmazonAnthropicClaudeMessagesConfig(
),
custom_llm_provider="bedrock",
is_mid_conversation_output_config_used=is_mid_conversation_output_config_used,
is_thinking_display_updates_used=anthropic_model_info.is_thinking_display_updates_used(
anthropic_messages_request.get("thinking")
),
)
beta_set.update(auto_betas)
@ -664,6 +678,8 @@ class AmazonAnthropicClaudeMessagesConfig(
litellm_params: GenericLiteLLMParams,
headers: dict,
) -> dict:
requested_thinking: Final = anthropic_messages_optional_request_params.get("thinking")
requested_display_updates: Final = AnthropicModelInfo().is_thinking_display_updates_used(requested_thinking)
self._clamp_adaptive_reasoning_effort_for_bedrock(
model=model,
optional_params=anthropic_messages_optional_request_params,
@ -676,6 +692,14 @@ class AmazonAnthropicClaudeMessagesConfig(
litellm_params=litellm_params,
headers=headers,
)
translated_thinking: Final = anthropic_messages_request.get("thinking")
if (
requested_display_updates
and isinstance(translated_thinking, dict)
and translated_thinking.get("type") == "adaptive"
):
thinking_with_display: Final[AnthropicThinkingParam] = {"type": "adaptive", "display": "updates"}
anthropic_messages_request["thinking"] = thinking_with_display
self._normalize_system_role_messages(anthropic_messages_request, model=model)
#########################################################
############## BEDROCK Invoke SPECIFIC TRANSFORMATION ###

View file

@ -733,7 +733,7 @@ ANTHROPIC_API_ONLY_HEADERS: Final = { # fails if calling anthropic on vertex ai
class AnthropicThinkingParam(TypedDict, total=False):
type: ReadOnly[Literal["enabled", "adaptive", "disabled"]]
budget_tokens: int
display: ReadOnly[Literal["summarized", "omitted"]]
display: ReadOnly[Literal["summarized", "omitted", "updates"]]
class ANTHROPIC_HOSTED_TOOLS(str, Enum):
@ -776,6 +776,8 @@ ANTHROPIC_EFFORT_BETA_HEADER: Final = "effort-2025-11-24"
ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER: Final = "mid-conversation-output-config-2026-07-01"
ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER: Final = "thinking-display-updates-2026-08-18"
ANTHROPIC_FINE_GRAINED_TOOL_STREAMING_BETA_HEADER: Final = "fine-grained-tool-streaming-2025-05-14"
# OAuth constants

View file

@ -130,3 +130,23 @@ def test_json_provider_passthrough_adds_per_turn_control_beta():
)
assert PER_TURN_CONTROL in _betas(headers)
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
@pytest.mark.parametrize("explicit_beta", (False, True))
def test_native_messages_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
from typing import Final
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
headers, _ = AnthropicMessagesConfig().validate_anthropic_messages_environment(
headers={"anthropic-beta": beta} if explicit_beta else {},
model="claude-opus-5",
messages=[{"role": "user", "content": "Reply with OK"}],
optional_params={"thinking": {"type": "adaptive", "display": display}} if display else {},
litellm_params={},
api_key="sk-ant-test",
)
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(display == "updates" or explicit_beta)

View file

@ -1984,7 +1984,6 @@ class TestClaudeOpus48AdaptiveThinking:
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
@pytest.mark.parametrize(
"model",
[
@ -2376,3 +2375,78 @@ def test_validate_environment_adds_mid_conversation_output_config_beta(
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(nested_output_config or explicit_beta)
assert headers["x-api-key"] == FAKE_REGULAR_KEY
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
@pytest.mark.parametrize("explicit_beta", (False, True))
def test_validate_environment_adds_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
headers: Final = AnthropicModelInfo().validate_environment(
headers={"anthropic-beta": beta} if explicit_beta else {},
model="claude-opus-5",
messages=[{"role": "user", "content": "Reply with OK"}],
optional_params={"thinking": {"type": "adaptive", "display": display}} if display else {},
litellm_params={},
api_key=FAKE_REGULAR_KEY,
)
assert headers.get("anthropic-beta", "").split(",").count(beta) == int(display == "updates" or explicit_beta)
assert headers["x-api-key"] == FAKE_REGULAR_KEY
@pytest.mark.parametrize(
("thinking", "expected"),
(
(None, False),
({}, False),
("updates", False),
({"display": "updates"}, False),
({"type": "disabled", "display": "updates"}, False),
({"type": "enabled", "display": "updates", "budget_tokens": 1024}, True),
),
)
def test_thinking_display_beta_requires_active_thinking(thinking: object, expected: bool) -> None:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
headers: Final = AnthropicModelInfo().validate_environment(
headers={},
model="claude-opus-5",
messages=[{"role": "user", "content": "Reply with OK"}],
optional_params={"thinking": thinking},
litellm_params={},
api_key=FAKE_REGULAR_KEY,
)
assert (ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in headers.get("anthropic-beta", "").split(",")) is expected
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize(
("display", "expected_thinking"),
(
("summarized", {"type": "adaptive", "display": "summarized"}),
("omitted", {"type": "adaptive", "display": "omitted"}),
("updates", {"type": "adaptive"}),
),
)
def test_shared_legacy_thinking_translation_preserves_supported_display(
display: str, expected_thinking: dict[str, str]
) -> None:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
optional_params: Final = {
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": display},
}
AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model(
model="claude-opus-5",
optional_params=optional_params,
custom_llm_provider="azure_ai",
)
assert optional_params["thinking"] == expected_thinking

View file

@ -1,4 +1,3 @@
import asyncio
import base64
import copy
import json
@ -12,14 +11,12 @@ import pytest
# Ensure the project root is on the import path so `litellm` can be imported when
# tests are executed from any working directory.
import litellm
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeConfig,
)
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
ONE_PIXEL_PNG = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
)
@ -93,9 +90,7 @@ def local_beta_headers_config(monkeypatch):
def test_get_supported_params_thinking():
config = AmazonAnthropicClaudeConfig()
params = config.get_supported_openai_params(
model="anthropic.claude-sonnet-4-20250514-v1:0"
)
params = config.get_supported_openai_params(model="anthropic.claude-sonnet-4-20250514-v1:0")
assert "thinking" in params
@ -148,53 +143,23 @@ def test_aws_params_filtered_from_request_body():
result_json = json.dumps(result)
# Verify AWS authentication params are NOT in the request body
assert (
"aws_access_key_id" not in result_json
), "AWS access key should not be in request body"
assert (
"aws_secret_access_key" not in result_json
), "AWS secret key should not be in request body"
assert (
"aws_session_token" not in result_json
), "AWS session token should not be in request body"
assert (
"aws_region_name" not in result_json
), "AWS region should not be in request body"
assert (
"aws_role_name" not in result_json
), "AWS role name should not be in request body"
assert (
"aws_session_name" not in result_json
), "AWS session name should not be in request body"
assert (
"aws_profile_name" not in result_json
), "AWS profile name should not be in request body"
assert (
"aws_web_identity_token" not in result_json
), "AWS web identity token should not be in request body"
assert (
"aws_sts_endpoint" not in result_json
), "AWS STS endpoint should not be in request body"
assert (
"aws_bedrock_runtime_endpoint" not in result_json
), "AWS bedrock endpoint should not be in request body"
assert (
"aws_external_id" not in result_json
), "AWS external ID should not be in request body"
assert (
"aws_session_tags" not in result_json
), "AWS session tags should not be in request body"
assert "aws_access_key_id" not in result_json, "AWS access key should not be in request body"
assert "aws_secret_access_key" not in result_json, "AWS secret key should not be in request body"
assert "aws_session_token" not in result_json, "AWS session token should not be in request body"
assert "aws_region_name" not in result_json, "AWS region should not be in request body"
assert "aws_role_name" not in result_json, "AWS role name should not be in request body"
assert "aws_session_name" not in result_json, "AWS session name should not be in request body"
assert "aws_profile_name" not in result_json, "AWS profile name should not be in request body"
assert "aws_web_identity_token" not in result_json, "AWS web identity token should not be in request body"
assert "aws_sts_endpoint" not in result_json, "AWS STS endpoint should not be in request body"
assert "aws_bedrock_runtime_endpoint" not in result_json, "AWS bedrock endpoint should not be in request body"
assert "aws_external_id" not in result_json, "AWS external ID should not be in request body"
assert "aws_session_tags" not in result_json, "AWS session tags should not be in request body"
# Also check that the sensitive values themselves are not in the response
assert (
"AKIAIOSFODNN7EXAMPLE" not in result_json
), "AWS access key value leaked in request body"
assert (
"wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY" not in result_json
), "AWS secret key value leaked in request body"
assert (
"arn:aws:iam::123456789012:role/test-role" not in result_json
), "AWS role ARN leaked in request body"
assert "AKIAIOSFODNN7EXAMPLE" not in result_json, "AWS access key value leaked in request body"
assert "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY" not in result_json, "AWS secret key value leaked in request body"
assert "arn:aws:iam::123456789012:role/test-role" not in result_json, "AWS role ARN leaked in request body"
assert "test-session" not in result_json, "AWS session name leaked in request body"
# Verify normal params ARE still in the request body
@ -203,9 +168,7 @@ def test_aws_params_filtered_from_request_body():
assert result["top_p"] == 0.9, "top_p should be in request body"
# Verify Bedrock-specific params are added
assert (
result["anthropic_version"] == "bedrock-2023-05-31"
), "anthropic_version should be set"
assert result["anthropic_version"] == "bedrock-2023-05-31", "anthropic_version should be set"
assert "model" not in result, "model should be removed for Bedrock Invoke API"
assert "stream" not in result, "stream should be removed for Bedrock Invoke API"
@ -262,9 +225,7 @@ def test_output_format_conversion_to_inline_schema():
)
# Verify output_format was removed from the request
assert (
"output_format" not in result
), "output_format should be removed from request body"
assert "output_format" not in result, "output_format should be removed from request body"
# Verify the schema was added to the last user message content
assert "messages" in result
@ -415,9 +376,7 @@ def test_opus_4_5_model_detection():
]
for model in non_opus_4_5_models:
assert not config._is_claude_opus_4_5(
model
), f"Should not detect {model} as Opus 4.5"
assert not config._is_claude_opus_4_5(model), f"Should not detect {model} as Opus 4.5"
# def test_structured_outputs_beta_header_filtered_for_bedrock_invoke():
@ -595,9 +554,7 @@ def test_output_config_format_forwarded_for_bedrock_chat_invoke_request(local_mo
("anthropic.claude-opus-4-7", "xhigh"),
],
)
def test_output_config_effort_normalized_for_bedrock_chat_invoke_request(
model, expected_effort
):
def test_output_config_effort_normalized_for_bedrock_chat_invoke_request(model, expected_effort):
"""Bedrock Invoke chat path accepts ``xhigh`` and forwards the provider-safe effort."""
config = AmazonAnthropicClaudeConfig()
@ -668,9 +625,9 @@ def test_output_format_removed_from_bedrock_invoke_request():
)
# Verify output_format is not in the request
assert (
"output_format" not in result
), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
assert "output_format" not in result, (
f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
)
def test_bedrock_chat_invoke_forwards_output_config_format_natively(local_model_cost_map):
@ -866,7 +823,9 @@ async def test_bedrock_invoke_claude_async_completion_inlines_remote_images_off_
assert async_only_image_fetch.base64_png in captured["body"]
async def test_bedrock_invoke_claude_async_completion_inlines_document_url_sources_off_the_event_loop(async_only_image_fetch):
async def test_bedrock_invoke_claude_async_completion_inlines_document_url_sources_off_the_event_loop(
async_only_image_fetch,
):
pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf"
captured = {}
@ -958,6 +917,62 @@ def test_bedrock_chat_invoke_tool_search_beta_follows_model_map(
assert result.get("anthropic_beta") == expected_betas
def test_bedrock_chat_invoke_adds_thinking_display_updates_beta(
local_model_cost_map, local_beta_headers_config
) -> None:
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
config: Final = AmazonAnthropicClaudeConfig()
model: Final = "us.anthropic.claude-opus-5"
optional_params: Final = config.map_openai_params(
non_default_params={
"max_tokens": 512,
"thinking": {"type": "adaptive", "display": "updates"},
},
optional_params={},
model=model,
drop_params=False,
)
result: Final = config.transform_request(
model=model,
messages=[{"role": "user", "content": "Reply with OK"}],
optional_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
def test_bedrock_chat_invoke_preserves_display_when_translating_legacy_thinking(
local_model_cost_map, local_beta_headers_config
) -> None:
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
config: Final = AmazonAnthropicClaudeConfig()
model: Final = "us.anthropic.claude-opus-5"
optional_params: Final = config.map_openai_params(
non_default_params={
"max_tokens": 512,
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": "updates"},
},
optional_params={},
model=model,
drop_params=False,
)
result: Final = config.transform_request(
model=model,
messages=[{"role": "user", "content": "Reply with OK"}],
optional_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
FINE_GRAINED_TOOL_STREAMING_BETA: Final = "fine-grained-tool-streaming-2025-05-14"
EAGER_TOOL_SCHEMA: Final = {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}
@ -1015,7 +1030,10 @@ def test_bedrock_chat_invoke_eager_input_streaming_beta_not_duplicated_with_clie
def _mid_conversation_system_conversation() -> list[dict]:
return [
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
{
"role": "system",
"content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}],
},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
@ -1074,7 +1092,11 @@ def _preserved_thinking_turns(reminder_after_user: bool) -> tuple[list[dict], li
second_question = {"role": "user", "content": "Second question"}
second_turn = [second_question, reminder] if reminder_after_user else [reminder, second_question]
turn_n_plus_one = [*turn_n, _thinking_reply("First answer"), *second_turn]
turn_n_plus_two = [*turn_n_plus_one, _thinking_reply("Second answer"), {"role": "user", "content": "Third question"}]
turn_n_plus_two = [
*turn_n_plus_one,
_thinking_reply("Second answer"),
{"role": "user", "content": "Third question"},
]
return turn_n, turn_n_plus_one, turn_n_plus_two
@ -1102,7 +1124,11 @@ def test_chat_flagged_model_replays_a_byte_identical_prefix_around_a_mid_convers
request must be a byte-identical prefix of turn N+1's or the block is dropped."""
requests = [
AmazonAnthropicClaudeConfig().transform_request(
model="invoke/us.anthropic.claude-fable-5-1", messages=copy.deepcopy(turn), optional_params={}, litellm_params={}, headers={}
model="invoke/us.anthropic.claude-fable-5-1",
messages=copy.deepcopy(turn),
optional_params={},
litellm_params={},
headers={},
)
for turn in _preserved_thinking_turns(reminder_after_user)
]

View file

@ -5,33 +5,33 @@ import json
import os
import struct
import zlib
from collections.abc import AsyncIterator, Mapping, Sequence
from datetime import datetime
from types import SimpleNamespace
from collections.abc import AsyncIterator, Mapping, Sequence
from typing import Final
from unittest.mock import Mock
import httpx
import pytest
# Ensure the project root is on the import path so `litellm` can be imported when
# tests are executed from any working directory.
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.bedrock.common_utils import (
ensure_bedrock_anthropic_messages_tool_names,
normalize_custom_field_on_tools,
normalize_tool_input_schema_types_for_bedrock_invoke,
)
from litellm.constants import (
BEDROCK_MIN_THINKING_BUDGET_TOKENS,
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
# Ensure the project root is on the import path so `litellm` can be imported when
# tests are executed from any working directory.
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.anthropic.pass_through.messages.mid_conversation_system import (
as_system_content_blocks,
)
from litellm.llms.bedrock.common_utils import (
ensure_bedrock_anthropic_messages_tool_names,
normalize_custom_field_on_tools,
normalize_tool_input_schema_types_for_bedrock_invoke,
)
from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeMessagesConfig,
AmazonAnthropicClaudeMessagesStreamDecoder,
@ -54,9 +54,7 @@ async def test_bedrock_sse_wrapper_encodes_dict_chunks():
_dummy_stream(),
litellm_logging_obj=LiteLLMLoggingObj(
model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0",
messages=[
{"role": "user", "content": "Hello, can you tell me a short joke?"}
],
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
stream=True,
call_type="chat",
start_time=datetime.now(),
@ -233,9 +231,7 @@ async def test_bedrock_sse_wrapper_keeps_usage_in_message_start_and_message_delt
def test_chunk_parser_usage_transformation():
"""Ensure Bedrock invocation metrics are transformed to Anthropic usage keys."""
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0"
)
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-3-sonnet-20240229-v1:0")
chunk = {
"type": "message_delta",
@ -264,9 +260,7 @@ def test_chunk_parser_preserves_cache_usage_fields_with_invocation_metrics():
fields and cache tokens end up billed at $0.
"""
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
)
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
chunk = {
"type": "message_stop",
@ -292,9 +286,7 @@ def test_chunk_parser_preserves_cache_usage_fields_with_invocation_metrics():
def test_chunk_parser_maps_cache_token_counts_from_invocation_metrics():
"""Cache itemization inside invocationMetrics maps to Anthropic usage keys."""
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
)
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
chunk = {
"type": "message_stop",
@ -317,9 +309,7 @@ def test_chunk_parser_maps_cache_token_counts_from_invocation_metrics():
def test_chunk_parser_keeps_existing_token_counts_over_invocation_metrics():
"""Token counts reported in the chunk's own usage block win over invocationMetrics."""
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
)
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
chunk = {
"type": "message_stop",
@ -354,9 +344,7 @@ async def test_bedrock_sse_wrapper_preserves_cache_usage_with_invocation_metrics
final usage billed cache reads and writes at $0.
"""
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(
model="bedrock/invoke/anthropic.claude-sonnet-4-6"
)
decoder = AmazonAnthropicClaudeMessagesStreamDecoder(model="bedrock/invoke/anthropic.claude-sonnet-4-6")
cfg = AmazonAnthropicClaudeMessagesConfig()
raw_chunks = [
@ -566,11 +554,7 @@ def test_normalize_custom_field_on_tools():
assert request4["tools"] is None
# Case 5: an explicit top-level flag wins over a conflicting wrapped one
request5 = {
"tools": [
{"name": "Read", "defer_loading": False, "custom": {"defer_loading": True}}
]
}
request5 = {"tools": [{"name": "Read", "defer_loading": False, "custom": {"defer_loading": True}}]}
normalize_custom_field_on_tools(request5)
assert request5["tools"][0] == {"name": "Read", "defer_loading": False}
@ -591,9 +575,7 @@ def test_normalize_custom_field_on_tools():
assert request7["tools"] == [{"name": "Read"}, {"name": "Write"}]
@pytest.mark.parametrize(
"deferred_marker", [{"custom": {"defer_loading": True}}, {"defer_loading": True}]
)
@pytest.mark.parametrize("deferred_marker", [{"custom": {"defer_loading": True}}, {"defer_loading": True}])
def test_bedrock_invoke_messages_transform_emits_top_level_defer_loading(
deferred_marker,
):
@ -726,9 +708,7 @@ def test_bedrock_invoke_messages_skips_thinking_injection_when_already_enabled(
"max_tokens": 32000,
"stream": False,
"thinking": {"type": "enabled", "budget_tokens": 2048},
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
result = cfg.transform_anthropic_messages_request(
model="global.anthropic.claude-sonnet-4-6-v1:0",
@ -830,9 +810,7 @@ def test_remove_ttl_from_cache_control_processes_tools(local_model_cost_map):
"messages": [],
}
cfg._remove_ttl_from_cache_control(
request, model="anthropic.claude-3-5-sonnet-20241022-v2:0"
)
cfg._remove_ttl_from_cache_control(request, model="anthropic.claude-3-5-sonnet-20241022-v2:0")
# Tool ttl should be stripped
assert "ttl" not in request["tools"][0]["cache_control"]
@ -868,9 +846,7 @@ def test_remove_ttl_from_cache_control_preserves_tools_ttl_for_claude_4_5(local_
],
}
cfg._remove_ttl_from_cache_control(
request, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0"
)
cfg._remove_ttl_from_cache_control(request, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0")
# Both tools and system should preserve ttl for Claude 4.5
assert request["tools"][0]["cache_control"]["ttl"] == "1h"
@ -954,9 +930,7 @@ def test_bedrock_messages_strips_output_config():
headers={},
)
assert "output_config" not in result, (
"output_config should be stripped for models that don't support it"
)
assert "output_config" not in result, "output_config should be stripped for models that don't support it"
assert result.get("max_tokens") == 4096
@ -989,9 +963,7 @@ def test_bedrock_messages_preserves_output_config_for_claude_4_6():
headers={},
)
assert "output_config" in result, (
"output_config should be preserved for supported models"
)
assert "output_config" in result, "output_config should be preserved for supported models"
assert result["output_config"] == {"effort": "high"}
assert result.get("max_tokens") == 4096
@ -1143,9 +1115,7 @@ def test_bedrock_messages_converts_output_config_format_to_inline_schema():
("anthropic.claude-opus-4-7", "xhigh"),
],
)
def test_bedrock_messages_normalizes_output_config_effort_for_opus(
model, expected_effort
):
def test_bedrock_messages_normalizes_output_config_effort_for_opus(model, expected_effort):
"""Bedrock /v1/messages accepts ``xhigh`` and forwards the provider-safe effort."""
from unittest.mock import patch
@ -1203,9 +1173,7 @@ def test_bedrock_messages_does_not_mutate_callers_messages_when_embedding_schema
headers={},
)
assert caller_messages == [
{"role": "user", "content": [{"type": "text", "text": "Hello"}]}
]
assert caller_messages == [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
assert caller_message == {
"role": "user",
"content": [{"type": "text", "text": "Hello"}],
@ -1521,9 +1489,7 @@ def test_bedrock_messages_strips_context_management():
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
result = cfg.transform_anthropic_messages_request(
@ -1534,9 +1500,7 @@ def test_bedrock_messages_strips_context_management():
headers={},
)
assert "context_management" not in result, (
"context_management should be stripped — Bedrock Invoke rejects it"
)
assert "context_management" not in result, "context_management should be stripped — Bedrock Invoke rejects it"
assert result.get("max_tokens") == 4096
@ -1661,7 +1625,9 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
["dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14", "interleaved-thinking-2025-05-14"],
ids=["client_sends_beta", "client_omits_beta"],
)
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(local_beta_headers_config, client_beta_header):
def test_bedrock_messages_forwards_safeguards_with_dangerous_tool_use_beta(
local_beta_headers_config, client_beta_header
):
"""
Claude Code's server-side auto-mode classifier sends `safeguards` alongside the
dangerous-tool-use-2026-09-03 beta. Bedrock Invoke accepts the pair, answers
@ -1769,12 +1735,8 @@ def test_bedrock_messages_filters_user_provided_unsupported_beta_header():
)
betas = result.get("anthropic_beta") or []
assert "advisor-tool-2026-03-01" not in betas, (
"user-provided beta not in the Bedrock mapping must be dropped"
)
assert "context-1m-2025-08-07" in betas, (
"user-provided beta that IS in the Bedrock mapping should survive"
)
assert "advisor-tool-2026-03-01" not in betas, "user-provided beta not in the Bedrock mapping must be dropped"
assert "context-1m-2025-08-07" in betas, "user-provided beta that IS in the Bedrock mapping should survive"
def test_bedrock_messages_renames_user_provided_aliased_beta_header():
@ -1802,9 +1764,7 @@ def test_bedrock_messages_renames_user_provided_aliased_beta_header():
assert "advanced-tool-use-2025-11-20" not in betas, (
"Anthropic-direct spelling should be rewritten, not forwarded verbatim"
)
assert "tool-search-tool-2025-10-19" in betas, (
"user-provided beta should be renamed to the Bedrock-side spelling"
)
assert "tool-search-tool-2025-10-19" in betas, "user-provided beta should be renamed to the Bedrock-side spelling"
@pytest.mark.asyncio
@ -2066,9 +2026,7 @@ async def test_unified_bedrock_messages_sse_usage_and_cost_claude_sonnet_46():
"global.anthropic.claude-fable-5",
],
)
def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models(
local_model_cost_map, model
):
def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models(local_model_cost_map, model):
"""clear_thinking_20251015 without a top-level ``thinking`` field must inject
``thinking.type=adaptive`` plus ``output_config.effort`` on adaptive-thinking
models (Opus 4.7/4.8, Fable 5). The legacy ``thinking.type=enabled`` shape is
@ -2078,9 +2036,7 @@ def test_bedrock_clear_thinking_injects_adaptive_with_effort_for_adaptive_models
cfg = AmazonAnthropicClaudeMessagesConfig()
request = {
"max_tokens": 32000,
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2103,9 +2059,7 @@ def test_bedrock_clear_thinking_converts_legacy_enabled_budget_to_effort():
"type": "enabled",
"budget_tokens": DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
},
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2123,10 +2077,7 @@ def test_resolve_clear_thinking_budget_tokens_honors_explicit_zero():
and only fall back to the minimum when the caller omits the budget."""
cfg = AmazonAnthropicClaudeMessagesConfig()
assert cfg._resolve_clear_thinking_budget_tokens(0) == 0
assert (
cfg._resolve_clear_thinking_budget_tokens(None)
== BEDROCK_MIN_THINKING_BUDGET_TOKENS
)
assert cfg._resolve_clear_thinking_budget_tokens(None) == BEDROCK_MIN_THINKING_BUDGET_TOKENS
assert cfg._resolve_clear_thinking_budget_tokens(12000) == 12000
@ -2136,9 +2087,7 @@ def test_bedrock_clear_thinking_keeps_enabled_for_non_adaptive_models():
cfg = AmazonAnthropicClaudeMessagesConfig()
request = {
"max_tokens": 32000,
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2163,9 +2112,7 @@ def test_bedrock_invoke_transform_emits_adaptive_thinking_for_opus_4_8():
optional_params = {
"max_tokens": 32000,
"stream": False,
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
result = cfg.transform_anthropic_messages_request(
@ -2202,9 +2149,7 @@ def test_bedrock_invoke_transform_normalizes_system_role_message_into_system():
assert all(m.get("role") != "system" for m in result["messages"])
assert result["messages"] == [{"role": "user", "content": "hi"}]
assert result["system"] == [
{"type": "text", "text": "You are a careful assistant."}
]
assert result["system"] == [{"type": "text", "text": "You are a careful assistant."}]
def test_bedrock_invoke_transform_merges_system_role_into_existing_system():
@ -2319,9 +2264,7 @@ def test_bedrock_invoke_transform_keeps_mid_conversation_system_role_in_place(lo
)
assert result["messages"] == messages
assert result["system"] == [
{"type": "text", "text": "Base.", "cache_control": {"type": "ephemeral"}}
]
assert result["system"] == [{"type": "text", "text": "Base.", "cache_control": {"type": "ephemeral"}}]
def test_bedrock_invoke_transform_hoists_only_leading_system_run(local_model_cost_map):
@ -2504,13 +2447,13 @@ def test_bedrock_invoke_transform_converted_system_carries_only_its_content(loca
assert result["messages"][2] == {
"role": "user",
"content": [
{
"type": "text",
"text": (
"Operator note (not from the user): the following was "
"originally a mid-conversation system-role reminder."
),
},
{
"type": "text",
"text": (
"Operator note (not from the user): the following was "
"originally a mid-conversation system-role reminder."
),
},
{"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"},
],
}
@ -2646,10 +2589,7 @@ def test_as_system_content_blocks_handles_each_shape():
def test_effort_from_thinking_budget_tiers(budget_tokens, expected_effort):
"""The budget -> effort mapping pins each tier boundary so a shifted threshold
is caught."""
assert (
AmazonAnthropicClaudeMessagesConfig._effort_from_thinking_budget(budget_tokens)
== expected_effort
)
assert AmazonAnthropicClaudeMessagesConfig._effort_from_thinking_budget(budget_tokens) == expected_effort
def test_inject_adaptive_thinking_preserves_existing_effort():
@ -2658,9 +2598,7 @@ def test_inject_adaptive_thinking_preserves_existing_effort():
cfg = AmazonAnthropicClaudeMessagesConfig()
request = {"output_config": {"effort": "max", "other": "keep"}}
cfg._inject_adaptive_thinking_for_clear_thinking(
request, budget_tokens=24000, model="us.anthropic.claude-fable-5"
)
cfg._inject_adaptive_thinking_for_clear_thinking(request, budget_tokens=24000, model="us.anthropic.claude-fable-5")
assert request["thinking"] == {"type": "adaptive"}
assert request["output_config"] == {"effort": "max", "other": "keep"}
@ -2673,9 +2611,7 @@ def test_bedrock_clear_thinking_noops_when_thinking_already_adaptive():
request = {
"max_tokens": 32000,
"thinking": {"type": "adaptive"},
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2695,9 +2631,7 @@ def test_bedrock_clear_thinking_replaces_disabled_thinking_on_adaptive_model():
request = {
"max_tokens": 32000,
"thinking": {"type": "disabled"},
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2717,9 +2651,7 @@ def test_bedrock_clear_thinking_leaves_enabled_thinking_on_non_adaptive_model():
request = {
"max_tokens": 32000,
"thinking": {"type": "enabled", "budget_tokens": 8000},
"context_management": {
"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]
},
"context_management": {"edits": [{"type": "clear_thinking_20251015", "keep": "all"}]},
}
changed = cfg._ensure_thinking_for_clear_thinking_context_management(
@ -2754,9 +2686,7 @@ def test_bedrock_messages_preserves_clear_tool_uses_context_management_and_adds_
messages = [{"role": "user", "content": [{"type": "text", "text": "Hi"}]}]
optional_params = {
"max_tokens": 4096,
"context_management": {
"edits": [{"type": "clear_tool_uses_20250919"}]
},
"context_management": {"edits": [{"type": "clear_tool_uses_20250919"}]},
}
result = cfg.transform_anthropic_messages_request(
@ -2767,12 +2697,11 @@ def test_bedrock_messages_preserves_clear_tool_uses_context_management_and_adds_
headers={},
)
assert result.get("context_management") == {
"edits": [{"type": "clear_tool_uses_20250919"}]
}, "clear_tool_uses_20250919 edit must reach Bedrock InvokeModel body"
assert result.get("context_management") == {"edits": [{"type": "clear_tool_uses_20250919"}]}, (
"clear_tool_uses_20250919 edit must reach Bedrock InvokeModel body"
)
assert "context-management-2025-06-27" in result.get("anthropic_beta", []), (
"context-management-2025-06-27 beta must reach the InvokeModel body so "
"the tool-call-clearing edit is accepted"
"context-management-2025-06-27 beta must reach the InvokeModel body so the tool-call-clearing edit is accepted"
)
@ -2849,9 +2778,9 @@ def test_bedrock_messages_filters_clear_thinking_keeps_clear_tool_uses(
cm = result.get("context_management")
assert cm is not None
assert [e.get("type") for e in cm["edits"]] == [
"clear_tool_uses_20250919"
], "clear_thinking_20251015 must still be stripped (LiteLLM-internal)"
assert [e.get("type") for e in cm["edits"]] == ["clear_tool_uses_20250919"], (
"clear_thinking_20251015 must still be stripped (LiteLLM-internal)"
)
betas = result.get("anthropic_beta", [])
assert "context-management-2025-06-27" in betas
@ -2992,9 +2921,7 @@ def test_bedrock_messages_tool_search_follows_claude_tool_search_rule(local_mode
assert cfg._supports_tool_search_on_bedrock(model) is expected
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
local_model_cost_map, monkeypatch
):
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(local_model_cost_map, monkeypatch):
"""The outbound thinking payload must follow the exact Bedrock cost-map entry.
Before threading the caller's provider through the capability probes, the probe
was pinned to ``"anthropic"``: the exact ``global.anthropic.claude-opus-4-8``
@ -3002,7 +2929,6 @@ def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
forced ``thinking.type='adaptive'`` even with ``supports_adaptive_thinking``
explicitly set to ``false`` on the entry."""
import litellm
from litellm.types.router import GenericLiteLLMParams
model = "global.anthropic.claude-opus-4-8"
@ -3404,22 +3330,14 @@ def test_bedrock_invoke_eager_input_streaming_beta_not_duplicated_with_client_he
def _bedrock_event_frame(payload: Mapping[str, object]) -> bytes:
def _header(name: str, value: str) -> bytes:
return (
bytes([len(name)])
+ name.encode()
+ bytes([7])
+ struct.pack(">H", len(value))
+ value.encode()
)
return bytes([len(name)]) + name.encode() + bytes([7]) + struct.pack(">H", len(value)) + value.encode()
headers: Final = (
_header(":message-type", "event")
+ _header(":event-type", "chunk")
+ _header(":content-type", "application/json")
)
body: Final = json.dumps(
{"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}
).encode()
body: Final = json.dumps({"bytes": base64.b64encode(json.dumps(payload).encode()).decode()}).encode()
prelude: Final = struct.pack(">II", 12 + len(headers) + len(body) + 4, len(headers))
prelude_crc: Final = struct.pack(">I", zlib.crc32(prelude))
message_crc: Final = struct.pack(">I", zlib.crc32(prelude + prelude_crc + headers + body))
@ -3547,3 +3465,65 @@ def test_bedrock_messages_removed_output_config_does_not_add_beta(explicit_beta:
assert result["messages"] == [{"role": "user", "content": "Reply with OK"}]
assert result.get("anthropic_beta", []).count(beta) == int(explicit_beta)
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
@pytest.mark.parametrize("display", (None, "summarized", "omitted", "updates"))
@pytest.mark.parametrize("explicit_beta", (False, True))
def test_bedrock_messages_thinking_display_updates_beta(display: str | None, explicit_beta: bool) -> None:
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
from litellm.types.router import GenericLiteLLMParams
beta: Final = ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
thinking: Final = {"type": "adaptive", "display": display} if display else None
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
model="eu.anthropic.claude-opus-5",
messages=[{"role": "user", "content": "Reply with OK"}],
anthropic_messages_optional_request_params={"max_tokens": 512, **({"thinking": thinking} if thinking else {})},
litellm_params=GenericLiteLLMParams(),
headers={"anthropic-beta": beta} if explicit_beta else {},
)
assert result.get("anthropic_beta", []).count(beta) == int(display == "updates" or explicit_beta)
assert result.get("thinking") == thinking
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
def test_bedrock_messages_preserves_display_when_translating_legacy_thinking() -> None:
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
from litellm.types.router import GenericLiteLLMParams
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
model="eu.anthropic.claude-opus-5",
messages=[{"role": "user", "content": "Reply with OK"}],
anthropic_messages_optional_request_params={
"max_tokens": 512,
"thinking": {"type": "enabled", "budget_tokens": 24000, "display": "updates"},
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])
@pytest.mark.usefixtures("local_model_cost_map", "local_beta_headers_config")
def test_bedrock_clear_thinking_preserves_display_updates() -> None:
from litellm.types.llms.anthropic import ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER
from litellm.types.router import GenericLiteLLMParams
result: Final = AmazonAnthropicClaudeMessagesConfig().transform_anthropic_messages_request(
model="us.anthropic.claude-opus-4-6",
messages=[{"role": "user", "content": "Reply with OK"}],
anthropic_messages_optional_request_params={
"max_tokens": 512,
"thinking": {"type": "enabled", "budget_tokens": 2048, "display": "updates"},
"context_management": {"edits": [{"type": "clear_thinking_20251015"}]},
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result.get("thinking") == {"type": "adaptive", "display": "updates"}
assert ANTHROPIC_THINKING_DISPLAY_UPDATES_BETA_HEADER in result.get("anthropic_beta", [])