From a62b79b90c0647dbbe250ba0fa5fc33379d9a8f3 Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Sun, 27 Sep 2026 17:31:25 +0800 Subject: [PATCH 1/8] fix(anthropic): preserve message cache control for list content --- .../prompt_templates/factory.py | 38 ++++++++++++++ litellm/llms/anthropic/chat/transformation.py | 11 ++++ ...llm_core_utils_prompt_templates_factory.py | 50 +++++++++++++++++++ .../test_anthropic_chat_transformation.py | 26 ++++++++++ 4 files changed, 125 insertions(+) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index c4e242fd360..cf5580b6692 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -2407,6 +2407,7 @@ def anthropic_messages_pt( ) = messages[msg_i] if user_message_types_block["role"] == "user": if isinstance(user_message_types_block["content"], list): + user_content_start: Final = len(user_content) for m in user_message_types_block["content"]: if m.get("type", "") == "image_url": m = cast(ChatCompletionImageObject, m) @@ -2478,6 +2479,16 @@ def anthropic_messages_pt( _file_content_element, ) ) + if ( + len(user_content) > user_content_start + and user_message_types_block.get("cache_control") is not None + and not any( + m.get("cache_control") is not None + for m in user_message_types_block["content"] + if isinstance(m, dict) + ) + ): + add_cache_control_to_content(user_content[-1], dict(user_message_types_block)) elif isinstance(user_message_types_block["content"], str): _anthropic_content_text_element: AnthropicMessagesTextParam = { "type": "text", @@ -2680,6 +2691,7 @@ def anthropic_messages_pt( ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR assistant_content.extend(thinking_blocks) if _content_is_list and _content_list is not None: + assistant_content_start: Final = len(assistant_content) for m in _content_list: if not isinstance(m, dict): continue @@ -2710,6 +2722,12 @@ def anthropic_messages_pt( # Pass through as-is since these are Anthropic-native content types elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): assistant_content.append(m) + if ( + len(assistant_content) > assistant_content_start + and assistant_content_block.get("cache_control") is not None + and not any(m.get("cache_control") is not None for m in _content_list if isinstance(m, dict)) + ): + add_cache_control_to_content(assistant_content[-1], dict(assistant_content_block)) elif ( "content" in assistant_content_block and isinstance(assistant_content_block["content"], str) @@ -4417,6 +4435,16 @@ class BedrockConverseMessagesProcessor: ) if _cache_point_block is not None: _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) user_content.extend(_parts) elif message_block["content"] and isinstance(message_block["content"], str): _part = BedrockContentBlock(text=messages[msg_i]["content"]) @@ -4790,6 +4818,16 @@ def _bedrock_converse_messages_pt( ) if _cache_point_block is not None: _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) user_content.extend(_parts) elif message_block["content"] and isinstance(message_block["content"], str): _part = BedrockContentBlock(text=messages[msg_i]["content"]) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 3bffee48d6a..24902c87ac2 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1719,6 +1719,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): anthropic_system_message_content["cache_control"] = system_message_block["cache_control"] anthropic_system_message_list.append(anthropic_system_message_content) elif isinstance(message["content"], list): + content_start: Final = len(anthropic_system_message_list) for _content in message["content"]: # Skip empty text blocks - Anthropic API raises errors for empty text text_value = _content.get("text") @@ -1739,6 +1740,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): anthropic_system_message_content["cache_control"] = _content["cache_control"] anthropic_system_message_list.append(anthropic_system_message_content) + if ( + len(anthropic_system_message_list) > content_start + and system_message_block.get("cache_control") is not None + and not any( + content.get("cache_control") is not None + for content in message["content"] + if isinstance(content, dict) + ) + ): + anthropic_system_message_list[-1]["cache_control"] = system_message_block["cache_control"] if len(system_prompt_indices) > 0: for idx in reversed(system_prompt_indices): diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index 1e12a973cdb..937ed191c97 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -30,6 +30,56 @@ from litellm.litellm_core_utils.prompt_templates.factory import ( from litellm.types.llms.openai import ChatCompletionToolMessage +@pytest.mark.parametrize("role", ["user", "assistant"]) +@pytest.mark.parametrize("block_cache_control", [False, True]) +def test_anthropic_list_content_preserves_message_cache_control(role, block_cache_control): + message_cache_control: Final = {"type": "ephemeral"} + block_cache: Final = {"type": "ephemeral", "ttl": "1h"} + content: Final = [ + {"type": "text", "text": "first", **({"cache_control": block_cache} if block_cache_control else {})}, + {"type": "text", "text": "second"}, + ] + message: Final = {"role": role, "content": content, "cache_control": message_cache_control} + messages: Final = [message] if role == "user" else [{"role": "user", "content": "hi"}, message] + + result: Final = anthropic_messages_pt(messages=messages, model="claude-sonnet-4-6", llm_provider="anthropic") + + expected: Final = [ + {"type": "text", "text": "first", **({"cache_control": block_cache} if block_cache_control else {})}, + { + "type": "text", + "text": "second", + **({} if block_cache_control else {"cache_control": message_cache_control}), + }, + ] + assert result[-1]["content"] == expected + + +@pytest.mark.asyncio +@pytest.mark.parametrize("block_cache_control", [False, True]) +async def test_bedrock_list_content_preserves_message_cache_control(block_cache_control): + content: Final = [ + {"type": "text", "text": "first", **({"cache_control": {"type": "ephemeral"}} if block_cache_control else {})}, + {"type": "text", "text": "second"}, + ] + messages: Final = [{"role": "user", "content": content, "cache_control": {"type": "ephemeral"}}] + model: Final = "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + + sync_result: Final = _bedrock_converse_messages_pt(messages=messages, model=model, llm_provider="bedrock") + async_result: Final = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, model=model, llm_provider="bedrock" + ) + + cache_point: Final = {"cachePoint": {"type": "default"}} + expected: Final = ( + [{"text": "first"}, cache_point, {"text": "second"}] + if block_cache_control + else [{"text": "first"}, {"text": "second"}, cache_point] + ) + assert sync_result[0]["content"] == expected + assert async_result[0]["content"] == expected + + def _get_gemini_function_response_inline_data_parts(result): assert isinstance(result, list), "expected Gemini parts list" assert len(result) == 1, "multimodal function responses should stay in one part" diff --git a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py index 332153b4c7d..659721274fa 100644 --- a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -2383,6 +2383,32 @@ def test_translate_system_message_preserves_cache_control(): assert result[0]["cache_control"] == {"type": "ephemeral"} +@pytest.mark.parametrize("block_cache_control", [False, True]) +def test_translate_system_list_content_preserves_message_cache_control(block_cache_control): + message_cache_control: Final = {"type": "ephemeral"} + block_cache: Final = {"type": "ephemeral", "ttl": "1h"} + content: Final = [ + {"type": "text", "text": "first", **({"cache_control": block_cache} if block_cache_control else {})}, + {"type": "text", "text": "second"}, + ] + messages: Final = [ + {"role": "system", "content": content, "cache_control": message_cache_control}, + {"role": "user", "content": "hi"}, + ] + + result: Final = AnthropicConfig().translate_system_message(messages) + + expected: Final = [ + {"type": "text", "text": "first", **({"cache_control": block_cache} if block_cache_control else {})}, + { + "type": "text", + "text": "second", + **({} if block_cache_control else {"cache_control": message_cache_control}), + }, + ] + assert result == expected + + # ============ Dynamic max_tokens Tests ============ From cb8d5f42d8554642d615781658b45d9e872b63fa Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Sun, 27 Sep 2026 17:57:09 +0800 Subject: [PATCH 2/8] fix(anthropic): satisfy basedpyright loop binding rules --- litellm/litellm_core_utils/prompt_templates/factory.py | 4 ++-- litellm/llms/anthropic/chat/transformation.py | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index cf5580b6692..725da3cd7a7 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -2407,7 +2407,7 @@ def anthropic_messages_pt( ) = messages[msg_i] if user_message_types_block["role"] == "user": if isinstance(user_message_types_block["content"], list): - user_content_start: Final = len(user_content) + user_content_start: int = len(user_content) for m in user_message_types_block["content"]: if m.get("type", "") == "image_url": m = cast(ChatCompletionImageObject, m) @@ -2691,7 +2691,7 @@ def anthropic_messages_pt( ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR assistant_content.extend(thinking_blocks) if _content_is_list and _content_list is not None: - assistant_content_start: Final = len(assistant_content) + assistant_content_start: int = len(assistant_content) for m in _content_list: if not isinstance(m, dict): continue diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 24902c87ac2..8ec20eff068 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1719,7 +1719,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): anthropic_system_message_content["cache_control"] = system_message_block["cache_control"] anthropic_system_message_list.append(anthropic_system_message_content) elif isinstance(message["content"], list): - content_start: Final = len(anthropic_system_message_list) + content_start: int = len(anthropic_system_message_list) for _content in message["content"]: # Skip empty text blocks - Anthropic API raises errors for empty text text_value = _content.get("text") From 7bb44abdcc078ea27eb79eeb34c97acab0c10c69 Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Sun, 27 Sep 2026 18:38:10 +0800 Subject: [PATCH 3/8] fix(anthropic): address cache breakpoint review findings --- .../prompt_templates/factory.py | 76 ++++++++++--- litellm/llms/anthropic/chat/transformation.py | 3 +- ...llm_core_utils_prompt_templates_factory.py | 105 ++++++++++++++++++ .../test_anthropic_chat_transformation.py | 18 +++ 4 files changed, 185 insertions(+), 17 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 725da3cd7a7..b3a300bbe46 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -2483,12 +2483,18 @@ def anthropic_messages_pt( len(user_content) > user_content_start and user_message_types_block.get("cache_control") is not None and not any( - m.get("cache_control") is not None - for m in user_message_types_block["content"] - if isinstance(m, dict) + "cache_control" in content and content["cache_control"] is not None + for content in user_content[user_content_start:] ) ): - add_cache_control_to_content(user_content[-1], dict(user_message_types_block)) + cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) + add_cache_control_to_content( + anthropic_content_element=cast( + dict[str, object], cached_user_content + ), # cast-ok: copied TypedDict is a dict + original_content_element=user_message_types_block, + ) + user_content[-1] = cached_user_content elif isinstance(user_message_types_block["content"], str): _anthropic_content_text_element: AnthropicMessagesTextParam = { "type": "text", @@ -2549,6 +2555,12 @@ def anthropic_messages_pt( _has_server_tool_calls = True break + _content_is_list: bool = "content" in assistant_content_block and isinstance( + assistant_content_block["content"], list + ) + _content_list = assistant_content_block.get("content") if _content_is_list else None + assistant_content_start: int = len(assistant_content) + if ( thinking_blocks is not None and _has_server_tool_calls @@ -2672,10 +2684,6 @@ def anthropic_messages_pt( # duplication and preserve the original interleaved order. # Fixes the gap where list-content messages bypass INTERLEAVED # MODE and still get thinking blocks prepended out of order. - _content_is_list = "content" in assistant_content_block and isinstance( - assistant_content_block["content"], list - ) - _content_list = assistant_content_block.get("content") if _content_is_list else None _list_has_thinking = False if _content_is_list and _content_list is not None: for _item in _content_list: @@ -2691,7 +2699,7 @@ def anthropic_messages_pt( ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR assistant_content.extend(thinking_blocks) if _content_is_list and _content_list is not None: - assistant_content_start: int = len(assistant_content) + assistant_content_start = len(assistant_content) for m in _content_list: if not isinstance(m, dict): continue @@ -2722,12 +2730,6 @@ def anthropic_messages_pt( # Pass through as-is since these are Anthropic-native content types elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): assistant_content.append(m) - if ( - len(assistant_content) > assistant_content_start - and assistant_content_block.get("cache_control") is not None - and not any(m.get("cache_control") is not None for m in _content_list if isinstance(m, dict)) - ): - add_cache_control_to_content(assistant_content[-1], dict(assistant_content_block)) elif ( "content" in assistant_content_block and isinstance(assistant_content_block["content"], str) @@ -2782,6 +2784,24 @@ def anthropic_messages_pt( if assistant_function_call is not None: assistant_content.extend(convert_function_to_anthropic_tool_invoke(assistant_function_call)) + if ( + _content_is_list + and len(assistant_content) > assistant_content_start + and assistant_content_block.get("cache_control") is not None + and not any( + "cache_control" in content and content["cache_control"] is not None + for content in assistant_content[assistant_content_start:] + ) + ): + cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) + add_cache_control_to_content( + anthropic_content_element=cast( + dict[str, object], cached_assistant_content + ), # cast-ok: copied TypedDict is a dict + original_content_element=assistant_content_block, + ) + assistant_content[-1] = cached_assistant_content + msg_i += 1 if assistant_content: @@ -4530,6 +4550,7 @@ class BedrockConverseMessagesProcessor: message=messages[msg_i], assistant_continue_message=assistant_continue_message, ) + assistant_message_content_start: int = len(assistant_content) _assistant_content = assistant_message_block.get("content", None) thinking_blocks = cast( list[ChatCompletionThinkingBlock] | None, @@ -4598,6 +4619,18 @@ class BedrockConverseMessagesProcessor: if _tool_calls: assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + msg_i += 1 assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") @@ -4915,6 +4948,7 @@ def _bedrock_converse_messages_pt( message=messages[msg_i], assistant_continue_message=assistant_continue_message, ) + assistant_message_content_start: int = len(assistant_content) _assistant_content = assistant_message_block.get("content", None) thinking_blocks = cast( list[ChatCompletionThinkingBlock] | None, @@ -4984,6 +5018,18 @@ def _bedrock_converse_messages_pt( if _tool_calls: assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + msg_i += 1 assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 8ec20eff068..a83f9ec5a54 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1745,8 +1745,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): and system_message_block.get("cache_control") is not None and not any( content.get("cache_control") is not None - for content in message["content"] - if isinstance(content, dict) + for content in anthropic_system_message_list[content_start:] ) ): anthropic_system_message_list[-1]["cache_control"] = system_message_block["cache_control"] diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index 937ed191c97..b310080687b 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -55,6 +55,73 @@ def test_anthropic_list_content_preserves_message_cache_control(role, block_cach assert result[-1]["content"] == expected +@pytest.mark.parametrize("call_field", ["tool_calls", "function_call"]) +def test_anthropic_list_message_cache_control_follows_tool_calls(call_field): + cache_control: Final = {"type": "ephemeral"} + call: Final = ( + [{"id": "call_1", "type": "function", "function": {"name": "lookup", "arguments": "{}"}}] + if call_field == "tool_calls" + else {"name": "lookup", "arguments": "{}"} + ) + messages: Final = [ + {"role": "user", "content": "hi"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "checking"}], + "cache_control": cache_control, + call_field: call, + }, + ] + + result: Final = anthropic_messages_pt(messages=messages, model="claude-sonnet-4-6", llm_provider="anthropic") + blocks: Final = result[-1]["content"] + + assert blocks[0] == {"type": "text", "text": "checking"} + assert blocks[-1]["type"] == "tool_use" + assert blocks[-1]["cache_control"] == cache_control + + +@pytest.mark.parametrize("role", ["user", "assistant"]) +def test_anthropic_list_message_cache_control_preserves_caller_block(role): + block: Final = ( + {"type": "document", "source": {"type": "text", "media_type": "text/plain", "data": "document"}} + if role == "user" + else {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {}} + ) + messages: Final = [ + {"role": "user", "content": "hi"}, + {"role": role, "content": [block], "cache_control": {"type": "ephemeral"}}, + ] + + result: Final = anthropic_messages_pt(messages=messages, model="claude-sonnet-4-6", llm_provider="anthropic") + + assert "cache_control" not in block + assert result[-1]["content"][-1]["cache_control"] == {"type": "ephemeral"} + assert result[-1]["content"][-1] is not block + + +@pytest.mark.parametrize("role", ["user", "assistant"]) +def test_anthropic_skipped_list_block_does_not_suppress_message_cache_control(role): + skipped_block: Final = ( + {"type": "unknown", "cache_control": {"type": "ephemeral", "ttl": "1h"}} + if role == "user" + else {"type": "thinking", "thinking": "unsigned", "cache_control": {"type": "ephemeral", "ttl": "1h"}} + ) + messages: Final = [ + {"role": "user", "content": "hi"}, + { + "role": role, + "content": [skipped_block, {"type": "text", "text": "retained"}], + "cache_control": {"type": "ephemeral"}, + }, + ] + + result: Final = anthropic_messages_pt(messages=messages, model="claude-sonnet-4-6", llm_provider="anthropic") + + assert result[-1]["content"][-1] == {"type": "text", "text": "retained", "cache_control": {"type": "ephemeral"}} + assert all(block.get("type") != skipped_block["type"] for block in result[-1]["content"]) + + @pytest.mark.asyncio @pytest.mark.parametrize("block_cache_control", [False, True]) async def test_bedrock_list_content_preserves_message_cache_control(block_cache_control): @@ -80,6 +147,44 @@ async def test_bedrock_list_content_preserves_message_cache_control(block_cache_ assert async_result[0]["content"] == expected +@pytest.mark.asyncio +@pytest.mark.parametrize("block_cache_control", [False, True]) +@pytest.mark.parametrize("with_tool_calls", [False, True]) +async def test_bedrock_assistant_list_preserves_message_cache_control(block_cache_control, with_tool_calls): + content: Final = [ + { + "type": "text", + "text": "checking", + **({"cache_control": {"type": "ephemeral"}} if block_cache_control else {}), + } + ] + assistant_message: Final = { + "role": "assistant", + "content": content, + "cache_control": {"type": "ephemeral"}, + **( + {"tool_calls": [{"id": "call_1", "type": "function", "function": {"name": "lookup", "arguments": "{}"}}]} + if with_tool_calls + else {} + ), + } + messages: Final = [{"role": "user", "content": "hi"}, assistant_message] + model: Final = "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + + sync_result: Final = _bedrock_converse_messages_pt(messages=messages, model=model, llm_provider="bedrock") + async_result: Final = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, model=model, llm_provider="bedrock" + ) + + for result in (sync_result, async_result): + blocks = result[-1]["content"] + assert sum("cachePoint" in block for block in blocks) == 1 + if not block_cache_control: + assert blocks[-1] == {"cachePoint": {"type": "default"}} + if with_tool_calls: + assert "toolUse" in blocks[-2] + + def _get_gemini_function_response_inline_data_parts(result): assert isinstance(result, list), "expected Gemini parts list" assert len(result) == 1, "multimodal function responses should stay in one part" diff --git a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py index 659721274fa..7fdf8032dba 100644 --- a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -2409,6 +2409,24 @@ def test_translate_system_list_content_preserves_message_cache_control(block_cac assert result == expected +def test_translate_system_skipped_list_block_does_not_suppress_message_cache_control(): + messages: Final = [ + { + "role": "system", + "content": [ + {"type": "text", "text": "", "cache_control": {"type": "ephemeral", "ttl": "1h"}}, + {"type": "text", "text": "retained"}, + ], + "cache_control": {"type": "ephemeral"}, + }, + {"role": "user", "content": "hi"}, + ] + + result: Final = AnthropicConfig().translate_system_message(messages) + + assert result == [{"type": "text", "text": "retained", "cache_control": {"type": "ephemeral"}}] + + # ============ Dynamic max_tokens Tests ============ From 9202ac133aa2d6f8720cfd374a8046cc984b1bad Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Sun, 27 Sep 2026 18:42:46 +0800 Subject: [PATCH 4/8] fix(anthropic): place cast suppressions on the cast lines --- .../prompt_templates/factory.py | 11516 ++++++++-------- 1 file changed, 5758 insertions(+), 5758 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index b3a300bbe46..6d7c3edd723 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -1,5758 +1,5758 @@ -import base64 -import copy -import hashlib -import itertools -import json -import mimetypes -import re -import xml.etree.ElementTree as ET -from collections.abc import Iterator, Mapping, Sequence -from enum import Enum -from types import MappingProxyType -from typing import Any, Final, TypeAlias, TypedDict, cast, overload - -from jinja2.sandbox import ImmutableSandboxedEnvironment - -import litellm -import litellm.types -import litellm.types.llms -from litellm import verbose_logger -from litellm._uuid import uuid -from litellm.constants import REDACTED_BY_LITELLM -from litellm.litellm_core_utils.prompt_templates.mid_conversation_system import anthropic_system_messages -from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get -from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client -from litellm.types.files import get_file_extension_from_mime_type -from litellm.types.llms.anthropic import * -from litellm.types.llms.bedrock import CachePointBlock -from litellm.types.llms.bedrock import MessageBlock as BedrockMessageBlock -from litellm.types.llms.custom_http import httpxSpecialProvider -from litellm.types.llms.ollama import OllamaVisionModelObject -from litellm.types.llms.openai import ( - AllMessageValues, - ChatCompletionAssistantMessage, - ChatCompletionAssistantToolCall, - ChatCompletionFileObject, - ChatCompletionFileObjectFile, - ChatCompletionFunctionMessage, - ChatCompletionImageObject, - ChatCompletionImageUrlObject, - ChatCompletionTextObject, - ChatCompletionToolCallFunctionChunk, - ChatCompletionToolMessage, - ChatCompletionUserMessage, - OpenAIMessageContentListBlock, -) -from litellm.types.llms.vertex_ai import FunctionCall as VertexFunctionCall -from litellm.types.llms.vertex_ai import FunctionResponse as VertexFunctionResponse -from litellm.types.llms.vertex_ai import PartType as VertexPartType -from litellm.types.utils import GenericImageParsingChunk - -from .common_utils import ( - convert_content_list_to_str, - infer_content_type_from_url_and_content, - is_non_content_values_set, - is_unsignable_thinking_block, - parse_tool_call_arguments, - salvage_concatenated_tool_arguments, -) -from .image_handling import convert_url_to_base64 - - -def default_pt(messages): - return " ".join(message["content"] for message in messages) - - -def prompt_injection_detection_default_pt(): - return """Detect if a prompt is safe to run. Return 'UNSAFE' if not.""" - - -BAD_MESSAGE_ERROR_STR: Final = "Invalid Message " - -# Separator used to embed Gemini thought signatures in tool call IDs -# See: https://ai.google.dev/gemini-api/docs/thought-signatures -THOUGHT_SIGNATURE_SEPARATOR: Final = "__thought__" - -# used to interweave user messages, to ensure user/assistant alternating -DEFAULT_USER_CONTINUE_MESSAGE: Final = { - "role": "user", - "content": "Please continue.", -} # similar to autogen. Only used if `litellm.modify_params=True`. - -DEFAULT_USER_CONTINUE_MESSAGE_TYPED: Final = ChatCompletionUserMessage( - role="user", - content="Please continue.", -) - -# used to interweave assistant messages, to ensure user/assistant alternating -DEFAULT_ASSISTANT_CONTINUE_MESSAGE: Final = ChatCompletionAssistantMessage( - role="assistant", - content=[ - { - "type": "text", - "text": "Please continue.", - } - ], -) # similar to autogen. Only used if `litellm.modify_params=True`. - - -def map_system_message_pt(messages: list) -> list: - """ - Convert 'system' message to 'user' message if provider doesn't support 'system' role. - - Enabled via `completion(...,supports_system_message=False)` - - If next message is a user message or assistant message -> merge system prompt into it - - if next message is system -> append a user message instead of the system message - """ - - new_messages: Final = [] - for i, m in enumerate(messages): - if m["role"] == "system": - if i < len(messages) - 1: # Not the last message - next_m = messages[i + 1] - next_role = next_m["role"] - if next_role == "user" or next_role == "assistant": # Next message is a user or assistant message - # Merge system prompt into the next message - next_m["content"] = m["content"] + " " + next_m["content"] - elif next_role == "system": # Next message is a system message - # Append a user message instead of the system message - new_message = {"role": "user", "content": m["content"]} - new_messages.append(new_message) - else: # Last message - new_message = {"role": "user", "content": m["content"]} - new_messages.append(new_message) - else: # Not a system message - new_messages.append(m) - - return new_messages - - -# alpaca prompt template - for models like mythomax, etc. -def alpaca_pt(messages): - prompt: Final = custom_prompt( - role_dict={ - "system": { - "pre_message": "### Instruction:\n", - "post_message": "\n\n", - }, - "user": { - "pre_message": "### Instruction:\n", - "post_message": "\n\n", - }, - "assistant": {"pre_message": "### Response:\n", "post_message": "\n\n"}, - }, - bos_token="", - eos_token="", - messages=messages, - ) - return prompt - - -# Llama2 prompt template -def llama_2_chat_pt(messages): - prompt: Final = custom_prompt( - role_dict={ - "system": { - "pre_message": "[INST] <>\n", - "post_message": "\n<>\n [/INST]\n", - }, - "user": { # follow this format https://github.com/facebookresearch/llama/blob/77062717054710e352a99add63d160274ce670c6/llama/generation.py#L348 - "pre_message": "[INST] ", - "post_message": " [/INST]\n", - }, - "assistant": { - "post_message": "\n" # follows this - https://replicate.com/blog/how-to-prompt-llama - }, - }, - messages=messages, - bos_token="", - eos_token="", - ) - return prompt - - -def convert_to_ollama_image(openai_image_url: str): - try: - if openai_image_url.startswith("http"): - openai_image_url = convert_url_to_base64(url=openai_image_url) - - if openai_image_url.startswith("data:image/"): - # Extract the base64 image data - base64_data = openai_image_url.split("data:image/")[1].split(";base64,")[1] - else: - base64_data = openai_image_url - - return base64_data - except Exception as e: - if "Error: Unable to fetch image from URL" in str(e): - raise e - raise Exception( - """Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{base64_image}". """ - ) - - -def _handle_ollama_system_message(messages: list, prompt: str, msg_i: int) -> tuple[str, int]: - system_content_str = "" - ## MERGE CONSECUTIVE SYSTEM CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - msg_content = convert_content_list_to_str(messages[msg_i]) - system_content_str += msg_content - - msg_i += 1 - - return system_content_str, msg_i - - -def ollama_pt( - model: str, messages: list -) -> ( - str | OllamaVisionModelObject -): # https://github.com/ollama/ollama/blob/af4cf55884ac54b9e637cd71dadfe9b7a5685877/docs/modelfile.md#template - user_message_types: Final = {"user", "tool", "function"} - msg_i = 0 - images: Final = [] - prompt = "" - while msg_i < len(messages): - init_msg_i = msg_i - user_content_str = "" - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - msg_content = messages[msg_i].get("content") - if msg_content: - if isinstance(msg_content, list): - for m in msg_content: - if m.get("type", "") == "image_url": - if isinstance(m["image_url"], str): - images.append(m["image_url"]) - elif isinstance(m["image_url"], dict): - images.append(m["image_url"]["url"]) - elif m.get("type", "") == "text": - user_content_str += m["text"] - else: - # Tool message content will always be a string - user_content_str += msg_content - - msg_i += 1 - - if user_content_str: - prompt += f"### User:\n{user_content_str}\n\n" - - system_content_str, msg_i = _handle_ollama_system_message(messages, prompt, msg_i) - if system_content_str: - prompt += f"### System:\n{system_content_str}\n\n" - - assistant_content_str = "" - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_content_str += convert_content_list_to_str(messages[msg_i]) - - tool_calls = messages[msg_i].get("tool_calls") - ollama_tool_calls = [] - if tool_calls: - for call in tool_calls: - call_id: str = call["id"] - function_name: str = call["function"]["name"] - arguments = json.loads(call["function"]["arguments"]) - - ollama_tool_calls.append( - { - "id": call_id, - "type": "function", - "function": { - "name": function_name, - "arguments": arguments, - }, - } - ) - - if ollama_tool_calls: - assistant_content_str += f"Tool Calls: {json.dumps(ollama_tool_calls, indent=2)}" - - msg_i += 1 - - if assistant_content_str: - prompt += f"### Assistant:\n{assistant_content_str}\n\n" - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider="ollama", - ) - - response_dict: Final[OllamaVisionModelObject] = { - "prompt": prompt, - "images": images, - } - - return response_dict - - -def mistral_instruct_pt(messages): - # Following the Mistral example's https://huggingface.co/docs/transformers/main/chat_templating - prompt: Final = custom_prompt( - initial_prompt_value="", - role_dict={ - "system": { - "pre_message": "[INST] \n", - "post_message": " [/INST]\n", - }, - "user": {"pre_message": "[INST] ", "post_message": " [/INST]\n"}, - "assistant": {"pre_message": " ", "post_message": " "}, - }, - final_prompt_value="", - messages=messages, - ) - return prompt - - -# Falcon prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 -def falcon_instruct_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += message["content"] - else: - prompt += message["role"] + ":" + message["content"].replace("\r\n", "\n").replace("\n\n", "\n") - prompt += "\n\n" - - return prompt - - -def falcon_chat_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "System: " + message["content"] - elif message["role"] == "assistant": - prompt += "Falcon: " + message["content"] - elif message["role"] == "user": - prompt += "User: " + message["content"] - - return prompt - - -# MPT prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 -def mpt_chat_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "<|im_start|>system" + message["content"] + "<|im_end|>" + "\n" - elif message["role"] == "assistant": - prompt += "<|im_start|>assistant" + message["content"] + "<|im_end|>" + "\n" - elif message["role"] == "user": - prompt += "<|im_start|>user" + message["content"] + "<|im_end|>" + "\n" - return prompt - - -# WizardCoder prompt template - https://huggingface.co/WizardLM/WizardCoder-Python-34B-V1.0#prompt-format -def wizardcoder_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += message["content"] + "\n\n" - elif message["role"] == "user": # map to 'Instruction' - prompt += "### Instruction:\n" + message["content"] + "\n\n" - elif message["role"] == "assistant": # map to 'Response' - prompt += "### Response:\n" + message["content"] + "\n\n" - return prompt - - -# Phind-CodeLlama prompt template - https://huggingface.co/Phind/Phind-CodeLlama-34B-v2#how-to-prompt-the-model -def phind_codellama_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "### System Prompt\n" + message["content"] + "\n\n" - elif message["role"] == "user": - prompt += "### User Message\n" + message["content"] + "\n\n" - elif message["role"] == "assistant": - prompt += "### Assistant\n" + message["content"] + "\n\n" - return prompt - - -def _render_chat_template(env, chat_template: str, bos_token: str, eos_token: str, messages: list) -> str: - """ - Shared template rendering logic for both sync and async hf_chat_template - - Args: - env: Jinja2 environment - chat_template: Chat template string - bos_token: Beginning of sequence token - eos_token: End of sequence token - messages: Messages to render - - Returns: - Rendered template string - """ - try: - template: Final = env.from_string(chat_template) - except Exception as e: - raise e - - def _is_system_in_template(): - try: - # Try rendering the template with a system message - template.render( - messages=[{"role": "system", "content": "test"}], - eos_token="", - bos_token="", - ) - return True - # This will be raised if Jinja attempts to render the system message and it can't - except Exception: - return False - - try: - rendered_text = "" - # Render the template with the provided values - if _is_system_in_template(): - rendered_text = template.render( - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - add_generation_prompt=True, - ) - else: - # treat a system message as a user message, if system not in template - reformatted_messages: Final = [] - try: - for message in messages: - if message["role"] == "system": - reformatted_messages.append({"role": "user", "content": message["content"]}) - else: - reformatted_messages.append(message) - rendered_text = template.render( - bos_token=bos_token, - eos_token=eos_token, - messages=reformatted_messages, - add_generation_prompt=True, - ) - except Exception as e: - if "Conversation roles must alternate user/assistant" in str(e): - # reformat messages to ensure user/assistant are alternating - new_messages: Final = [] - for i in range(len(reformatted_messages) - 1): - new_messages.append(reformatted_messages[i]) - if reformatted_messages[i]["role"] == reformatted_messages[i + 1]["role"]: - if reformatted_messages[i]["role"] == "user": - new_messages.append({"role": "assistant", "content": ""}) - else: - new_messages.append({"role": "user", "content": ""}) - new_messages.append(reformatted_messages[-1]) - rendered_text = template.render(bos_token=bos_token, eos_token=eos_token, messages=new_messages) - - return rendered_text - except Exception as e: - raise Exception(f"Error rendering template - {e}") # don't use verbose_logger.exception, if exception is raised - - -async def _afetch_and_extract_template( - model: str, chat_template: str | None, get_config_fn, get_template_fn -) -> tuple[str, str, str]: - """ - Async version: Fetch template and tokens from HuggingFace. - - Returns: (chat_template, bos_token, eos_token) - """ - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _extract_token_value, - ) - - bos_token = "" - eos_token = "" - - if chat_template is None: - # Fetch or retrieve cached tokenizer config - if model in litellm.known_tokenizer_config: - tokenizer_config = litellm.known_tokenizer_config[model] - else: - tokenizer_config = await get_config_fn(hf_model_name=model) - litellm.known_tokenizer_config.update({model: tokenizer_config}) - - # Try to get chat template from tokenizer_config.json first - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - and "chat_template" in tokenizer_config["tokenizer"] - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - chat_template = tokenizer_data["chat_template"] - else: - # Fallback: Try to fetch chat template from separate .jinja file - template_result: Final = await get_template_fn(hf_model_name=model) - if template_result.get("status") == "success": - chat_template = template_result["chat_template"] - # Still try to get tokens from tokenizer_config if available - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - else: - raise Exception("No chat template found") - - return chat_template, bos_token, eos_token - - -def _fetch_and_extract_template( - model: str, chat_template: str | None, get_config_fn, get_template_fn -) -> tuple[str, str, str]: - """ - Sync version: Fetch template and tokens from HuggingFace. - - Returns: (chat_template, bos_token, eos_token) - """ - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _extract_token_value, - ) - - bos_token = "" - eos_token = "" - - if chat_template is None: - # Fetch or retrieve cached tokenizer config - if model in litellm.known_tokenizer_config: - tokenizer_config = litellm.known_tokenizer_config[model] - else: - tokenizer_config = get_config_fn(hf_model_name=model) - litellm.known_tokenizer_config.update({model: tokenizer_config}) - - # Try to get chat template from tokenizer_config.json first - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - and "chat_template" in tokenizer_config["tokenizer"] - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - chat_template = tokenizer_data["chat_template"] - else: - # Fallback: Try to fetch chat template from separate .jinja file - template_result: Final = get_template_fn(hf_model_name=model) - if template_result.get("status") == "success": - chat_template = template_result["chat_template"] - # Still try to get tokens from tokenizer_config if available - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - else: - raise Exception("No chat template found") - - return chat_template, bos_token, eos_token - - -async def ahf_chat_template(model: str, messages: list, chat_template: str | None = None): - """HuggingFace chat template (async version)""" - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _aget_chat_template_file, - _aget_tokenizer_config, - strftime_now, - ) - - env: Final = ImmutableSandboxedEnvironment() - env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") - env.globals["strftime_now"] = strftime_now - - template, bos_token, eos_token = await _afetch_and_extract_template( - model=model, - chat_template=chat_template, - get_config_fn=_aget_tokenizer_config, - get_template_fn=_aget_chat_template_file, - ) - return _render_chat_template( - env=env, - chat_template=template, - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - ) - - -def hf_chat_template(model: str, messages: list, chat_template: str | None = None): - """HuggingFace chat template (sync version)""" - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _get_chat_template_file, - _get_tokenizer_config, - strftime_now, - ) - - env: Final = ImmutableSandboxedEnvironment() - env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") - env.globals["strftime_now"] = strftime_now - - template, bos_token, eos_token = _fetch_and_extract_template( - model=model, - chat_template=chat_template, - get_config_fn=_get_tokenizer_config, - get_template_fn=_get_chat_template_file, - ) - return _render_chat_template( - env=env, - chat_template=template, - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - ) - - -def deepseek_r1_pt(messages): - return hf_chat_template(model="deepseek-r1/deepseek-r1-7b-instruct", messages=messages) - - -# Anthropic template -def claude_2_1_pt( - messages: list, -): # format - https://docs.anthropic.com/claude/docs/how-to-use-system-prompts - """ - Claude v2.1 allows system prompts (no Human: needed), but requires it be followed by Human: - - you can't just pass a system message - - you can't pass a system message and follow that with an assistant message - if system message is passed in, you can only do system, human, assistant or system, human - - if a system message is passed in and followed by an assistant message, insert a blank human message between them. - - Additionally, you can "put words in Claude's mouth" by ending with an assistant message. - See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth - """ - - class AnthropicConstants(Enum): - HUMAN_PROMPT = "\n\nHuman: " - AI_PROMPT = "\n\nAssistant: " - - prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "user": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{message['content']}" - elif message["role"] == "assistant": - if idx > 0 and messages[idx - 1]["role"] == "system": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}" # Insert a blank human message - prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" - if messages[-1]["role"] != "assistant": - prompt += f"{AnthropicConstants.AI_PROMPT.value}" # prompt must end with \"\n\nAssistant: " turn - return prompt - - -### IBM Granite - - -def ibm_granite_pt(messages: list): - """ - IBM's Granite models uses the template: - <|system|> {system_message} <|user|> {user_message} <|assistant|> {assistant_message} - - See: https://www.ibm.com/docs/en/watsonx-as-a-service?topic=solutions-supported-foundation-models - """ - return custom_prompt( - messages=messages, - role_dict={ - "system": { - "pre_message": "<|system|>\n", - "post_message": "\n", - }, - "user": { - "pre_message": "<|user|>\n", - # Assistant tag is needed in the prompt after the user message - # to avoid the model completing the users sentence before it answers - # https://www.ibm.com/docs/en/watsonx/w-and-w/2.0.x?topic=models-granite-13b-chat-v2-prompting-tips#chat - "post_message": "\n<|assistant|>\n", - }, - "assistant": { - "pre_message": "", - "post_message": "\n", - }, - }, - ).strip() - - -### ANTHROPIC ### - - -def anthropic_pt( - messages: list, -): # format - https://docs.anthropic.com/claude/reference/complete_post - """ - You can "put words in Claude's mouth" by ending with an assistant message. - See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth - """ - - class AnthropicConstants(Enum): - HUMAN_PROMPT = "\n\nHuman: " - AI_PROMPT = "\n\nAssistant: " - - prompt = "" - for idx, message in enumerate(messages): # needs to start with `\n\nHuman: ` and end with `\n\nAssistant: ` - if message["role"] == "user": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - else: - prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" - if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` - prompt = f"{AnthropicConstants.HUMAN_PROMPT.value}" + prompt - if messages[-1]["role"] != "assistant": - prompt += f"{AnthropicConstants.AI_PROMPT.value}" - return prompt - - -def construct_format_parameters_prompt(parameters: dict): - parameter_str = "\n" - for k, v in parameters.items(): - parameter_str += f"<{k}>" - parameter_str += f"{v}" - parameter_str += f"" - parameter_str += "\n" - return parameter_str - - -def construct_format_tool_for_claude_prompt(name, description, parameters): - constructed_prompt: Final = ( - "\n" - f"{name}\n" - "\n" - f"{description}\n" - "\n" - "\n" - f"{construct_format_parameters_prompt(parameters)}\n" - "\n" - "" - ) - return constructed_prompt - - -def construct_tool_use_system_prompt( - tools, -): # from https://github.com/anthropics/anthropic-cookbook/blob/main/function_calling/function_calling.ipynb - tool_str_list: Final = [] - for tool in tools: - tool_function = get_attribute_or_key(tool, "function") - tool_str = construct_format_tool_for_claude_prompt( - get_attribute_or_key(tool_function, "name"), - get_attribute_or_key(tool_function, "description", ""), - get_attribute_or_key(tool_function, "parameters", {}), - ) - tool_str_list.append(tool_str) - tool_use_system_prompt: Final = ( - "In this environment you have access to a set of tools you can use to answer the user's question.\n" - "\n" - "You may call them like this:\n" - "\n" - "\n" - "$TOOL_NAME\n" - "\n" - "<$PARAMETER_NAME>$PARAMETER_VALUE\n" - "...\n" - "\n" - "\n" - "\n" - "\n" - "Here are the tools available:\n" - "\n" + "\n".join([tool_str for tool_str in tool_str_list]) + "\n" - ) - return tool_use_system_prompt - - -def convert_generic_image_chunk_to_openai_image_obj( - image_chunk: GenericImageParsingChunk, -) -> str: - """ - Convert a generic image chunk to an OpenAI image object. - - Input: - GenericImageParsingChunk( - type="base64", - media_type="image/jpeg", - data="...", - ) - - Return: - "data:image/jpeg;base64,{base64_image}" - """ - media_type: Final = image_chunk["media_type"] - return "data:{};{},{}".format(media_type, image_chunk["type"], image_chunk["data"]) - - -def convert_to_anthropic_image_obj(openai_image_url: str, format: str | None) -> GenericImageParsingChunk: - """ - Input: - "image_url": "data:image/jpeg;base64,{base64_image}", - - Return: - "source": { - "type": "base64", - "media_type": "image/jpeg", - "data": {base64_image}, - } - """ - try: - if openai_image_url.startswith("http"): - openai_image_url = convert_url_to_base64(url=openai_image_url) - # Extract the media type and base64 data - media_type, base64_data = openai_image_url.split("data:")[1].split(";base64,") - - if format: - media_type = format - else: - media_type = media_type.replace("\\/", "/") - - return GenericImageParsingChunk( - type="base64", - media_type=media_type, - data=base64_data, - ) - except litellm.ImageFetchError: - raise - except Exception as e: - raise Exception( - f"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{{base64_image}}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']. Error: {e}""" - ) - - -def create_anthropic_image_param( - image_url_input: str | dict, - format: str | None = None, - is_bedrock_invoke: bool = False, -) -> AnthropicMessagesImageParam: - """ - Create an AnthropicMessagesImageParam from an image URL input. - - Supports both URL references (for HTTP/HTTPS URLs) and base64 encoding. - """ - # Extract URL and format from input - if isinstance(image_url_input, str): - image_url = image_url_input - else: - image_url = image_url_input.get("url", "") - if format is None: - format = image_url_input.get("format") - - # Check if the image URL is an HTTP/HTTPS URL - if image_url.startswith("http://") or image_url.startswith("https://"): - # For Bedrock invoke and Vertex AI Anthropic, always convert URLs to base64 - # as these providers don't support URL sources for images - if is_bedrock_invoke or image_url.startswith("http://"): - base64_url: Final = convert_url_to_base64(url=image_url) - image_chunk = convert_to_anthropic_image_obj(openai_image_url=base64_url, format=format) - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - else: - # HTTPS URL - pass directly for regular Anthropic - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSourceUrl( - type="url", - url=image_url, - ), - ) - else: - # Convert to base64 for data URIs or other formats - image_chunk = convert_to_anthropic_image_obj(openai_image_url=image_url, format=format) - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - - -# The following XML functions will be deprecated once JSON schema support is available on Bedrock and Vertex -# ------------------------------------------------------------------------------ -def convert_to_anthropic_tool_result_xml(message: dict) -> str: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - """ - - """ - Anthropic tool_results look like: - - [Successful results] - - - get_current_weather - - function result goes here - - - - - [Error results] - - - error message goes here - - - """ - name: Final = message.get("name") - content = message.get("content", "") - content = content.replace("<", "<").replace(">", ">").replace("&", "&") - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - anthropic_tool_result: Final = ( - "\n" - "\n" - f"{name}\n" - "\n" - f"{content}\n" - "\n" - "\n" - "" - ) - - return anthropic_tool_result - - -def convert_to_anthropic_tool_invoke_xml(tool_calls: list) -> str: - invokes = "" - for tool in tool_calls: - if get_attribute_or_key(tool, "type") != "function": - continue - - tool_function = get_attribute_or_key(tool, "function") - tool_name = get_attribute_or_key(tool_function, "name") - tool_arguments = get_attribute_or_key(tool_function, "arguments") - parsed_args = parse_tool_call_arguments( - tool_arguments, tool_name=tool_name, context="Anthropic XML tool invoke" - ) - if isinstance(parsed_args, dict): - parameters = "".join(f"<{param}>{val}\n" for param, val in parsed_args.items()) - else: - parameters = f"{parsed_args}\n" - invokes += f"\n{tool_name}\n\n{parameters}\n\n" - - anthropic_tool_invoke: Final = f"\n{invokes}" - - return anthropic_tool_invoke - - -def anthropic_messages_pt_xml(messages: list): - """ - format messages for anthropic - 1. Anthropic supports roles like "user" and "assistant", (here litellm translates system-> assistant) - 2. The first message always needs to be of role "user" - 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) - 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) - 5. System messages are a separate param to the Messages API (used for tool calling) - 6. Ensure we only accept role, content. (message.name is not supported) - """ - # add role=tool support to allow function call result/error submission - user_message_types: Final = {"user", "tool"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final = [] - msg_i = 0 - while msg_i < len(messages): - user_content = [] - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "image_url": - format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None - image_param = create_anthropic_image_param(m["image_url"], format=format) - # Convert to dict format for XML version - source = image_param["source"] - if isinstance(source, dict) and source.get("type") == "url": - # Type narrowing for URL source - url_source = cast(AnthropicContentParamSourceUrl, source) - user_content.append( - { - "type": "image", - "source": { - "type": "url", - "url": url_source["url"], - }, - } - ) - else: - # Type narrowing for base64 source - base64_source = cast(AnthropicContentParamSource, source) - user_content.append( - { - "type": "image", - "source": { - "type": "base64", - "media_type": base64_source["media_type"], - "data": base64_source["data"], - }, - } - ) - elif m.get("type", "") == "text": - user_content.append({"type": "text", "text": m["text"]}) - else: - # Tool message content will always be a string - user_content.append( - { - "type": "text", - "text": ( - convert_to_anthropic_tool_result_xml(messages[msg_i]) - if messages[msg_i]["role"] == "tool" - else messages[msg_i]["content"] - ), - } - ) - - msg_i += 1 - - if user_content: - new_messages.append({"role": "user", "content": user_content}) - - assistant_content = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_text = messages[msg_i].get("content") or "" # either string or none - if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion - assistant_text += convert_to_anthropic_tool_invoke_xml(messages[msg_i]["tool_calls"]) - - assistant_content.append({"type": "text", "text": assistant_text}) - msg_i += 1 - - if assistant_content: - new_messages.append({"role": "assistant", "content": assistant_content}) - - if not new_messages or new_messages[0]["role"] != "user": - if litellm.modify_params: - new_messages.insert(0, {"role": "user", "content": [{"type": "text", "text": "."}]}) - else: - raise Exception( - "Invalid first message. Should always start with 'role'='user' for Anthropic. System prompt is sent separately for Anthropic. set 'litellm.modify_params = True' or 'litellm_settings:modify_params = True' on proxy, to insert a placeholder user message - '.' as the first message, " - ) - - if new_messages[-1]["role"] == "assistant": - for content in new_messages[-1]["content"]: - if isinstance(content, dict) and content["type"] == "text": - content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message - - return new_messages - - -# ------------------------------------------------------------------------------ - - -def _azure_tool_call_invoke_helper( - function_call_params: ChatCompletionToolCallFunctionChunk, -) -> ChatCompletionToolCallFunctionChunk | None: - """ - Azure requires 'arguments' to be a string. - """ - if function_call_params.get("arguments") is None: - function_call_params["arguments"] = "" - return function_call_params - - -def _azure_image_url_helper(content: ChatCompletionImageObject): - if isinstance(content["image_url"], str): - content["image_url"] = {"url": content["image_url"]} - else: - content["image_url"] = cast( - ChatCompletionImageUrlObject, - {k: v for k, v in content["image_url"].items() if k != "format"}, - ) - - -def _azure_file_helper(content: ChatCompletionFileObject) -> None: - content["file"] = cast( - ChatCompletionFileObjectFile, - {k: v for k, v in content.get("file", {}).items() if k != "format"}, - ) - - -def convert_to_azure_openai_messages( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: - for m in messages: - if m["role"] == "assistant": - function_call = m.get("function_call", None) - if function_call is not None: - m["function_call"] = _azure_tool_call_invoke_helper(function_call) - - if m["role"] == "user" and isinstance(m.get("content"), list): - for content in m.get("content", []): - if isinstance(content, dict) and content.get("type") == "image_url": - _azure_image_url_helper(cast(ChatCompletionImageObject, content)) - elif isinstance(content, dict) and content.get("type") == "file": - _azure_file_helper(cast(ChatCompletionFileObject, content)) - return messages - - -# ------------------------------------------------------------------------------ - - -def infer_protocol_value( - value: object, -) -> Literal[ - "string_value", - "number_value", - "bool_value", - "struct_value", - "list_value", - "null_value", - "unknown", -]: - if value is None: - return "null_value" - if isinstance(value, int) or isinstance(value, float): - return "number_value" - if isinstance(value, str): - return "string_value" - if isinstance(value, bool): - return "bool_value" - if isinstance(value, dict): - return "struct_value" - if isinstance(value, list): - return "list_value" - - return "unknown" - - -def _gemini_tool_call_invoke_helper( - function_call_params: ChatCompletionToolCallFunctionChunk, - tool_call_id: str | None = None, -) -> VertexFunctionCall | None: - name: Final = function_call_params.get("name", "") or "" - arguments: Final = function_call_params.get("arguments", "") - if ( - isinstance(arguments, str) and len(arguments) == 0 - ): # pass empty dict, if arguments is empty string - prevents call from failing - arguments_dict = { - "type": "object", - } - else: - arguments_dict = json.loads(arguments) - function_call: Final = VertexFunctionCall( - name=name, - args=arguments_dict, - ) - if tool_call_id: - clean_id: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] - if clean_id: - function_call["id"] = clean_id - return function_call - - -def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: str | None) -> str: - """ - Embed thought signature into tool call ID for OpenAI client compatibility. - - Args: - tool_call_id: The tool call ID (e.g., "call_abc123...") - thought_signature: Base64-encoded signature from Gemini response - - Returns: - Tool call ID with embedded signature if present, otherwise original ID - Format: call___thought__ - - See: https://ai.google.dev/gemini-api/docs/thought-signatures - """ - if thought_signature: - return f"{tool_call_id}{THOUGHT_SIGNATURE_SEPARATOR}{thought_signature}" - return tool_call_id - - -def _get_thought_signature_from_tool(tool: dict) -> str | None: - """Extract thought signature from tool call's provider_specific_fields. - - If not provided try to extract thought signature from tool call id - - Checks both tool.provider_specific_fields and tool.function.provider_specific_fields. - Returns None when the tool call carries no signature; callers decide whether a - placeholder signature is needed. - """ - # First check tool's provider_specific_fields - provider_fields: Final = tool.get("provider_specific_fields") or {} - if isinstance(provider_fields, dict): - signature = provider_fields.get("thought_signature") - if signature: - return signature - - # Then check function's provider_specific_fields - function: Final = tool.get("function") - if function: - if isinstance(function, dict): - func_provider_fields: Final = function.get("provider_specific_fields") or {} - if isinstance(func_provider_fields, dict): - signature = func_provider_fields.get("thought_signature") - if signature: - return signature - elif hasattr(function, "provider_specific_fields") and function.provider_specific_fields: - if isinstance(function.provider_specific_fields, dict): - signature = function.provider_specific_fields.get("thought_signature") - if signature: - return signature - # Check if thought signature is embedded in tool call ID - tool_call_id: Final = tool.get("id") - if tool_call_id and THOUGHT_SIGNATURE_SEPARATOR in tool_call_id: - parts: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1) - if len(parts) == 2: - _, signature = parts - return signature - return None - - -def _get_dummy_thought_signature() -> str: - """Generate a dummy thought signature for models that require it. - - This is used when transferring conversation history from older models - (like gemini-2.5-flash) to gemini-3, which requires thought_signature - for strict validation. Google documents it as a last resort that "will - negatively impact model performance", so callers must only fall back to it - when no real signature is available. - - See: - https://ai.google.dev/gemini-api/docs/thought-signatures#faqs - https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures - """ - dummy_data: Final = b"skip_thought_signature_validator" - return base64.b64encode(dummy_data).decode("utf-8") - - -def convert_to_gemini_tool_call_invoke( - message: ChatCompletionAssistantMessage, - model: str | None = None, - forward_function_call_id: bool = False, -) -> list[VertexPartType]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - """ - Gemini tool call invokes: - { - "role": "model", - "parts": [ - { - "functionCall": { - "name": "get_current_weather", - "args": { - "unit": "fahrenheit", - "predicted_temperature": 45, - "location": "Boston, MA", - } - } - } - ] - } - """ - - """ - - json.load the arguments - """ - try: - _parts_list: Final[list[VertexPartType]] = [] - tool_calls: Final = message.get("tool_calls", None) - function_call: Final = message.get("function_call", None) - - from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( - VertexGeminiConfig, - ) - - needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model) - - if tool_calls is not None: - for tool in tool_calls: - if "function" in tool: - gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper( - function_call_params=tool["function"], - tool_call_id=(tool.get("id") if forward_function_call_id else None), - ) - if gemini_function_call is not None: - part_dict: VertexPartType = {"function_call": gemini_function_call} - thought_signature = _get_thought_signature_from_tool(dict(tool)) - # Gemini signs only the first functionCall part of a parallel batch, so scope the - # placeholder fallback to that part instead of fabricating one per sibling call: - # https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example - is_first_function_call = len(_parts_list) == 0 - if not thought_signature and is_first_function_call and needs_dummy_signature: - thought_signature = _get_dummy_thought_signature() - if thought_signature: - part_dict["thoughtSignature"] = thought_signature - - _parts_list.append(part_dict) - else: # don't silently drop params. Make it clear to user what's happening. - raise Exception( - f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {tool}" - ) - elif function_call is not None: - gemini_function_call = _gemini_tool_call_invoke_helper(function_call_params=function_call) - if gemini_function_call is not None: - part_dict_function: Final[VertexPartType] = {"function_call": gemini_function_call} - - # Extract thought signature from function_call's provider_specific_fields - thought_signature = None - provider_fields: Final = ( - function_call.get("provider_specific_fields") if isinstance(function_call, dict) else {} - ) - if isinstance(provider_fields, dict): - thought_signature = provider_fields.get("thought_signature") - - # If no signature found and model is gemini-3, use dummy signature - if not thought_signature and needs_dummy_signature: - thought_signature = _get_dummy_thought_signature() - - if thought_signature: - part_dict_function["thoughtSignature"] = thought_signature - - _parts_list.append(part_dict_function) - else: # don't silently drop params. Make it clear to user what's happening. - raise Exception( - f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {message}" - ) - return _parts_list - except Exception as e: - raise Exception(f"Unable to convert openai tool calls={message} to gemini tool calls. Received error={e}") - - -def convert_to_gemini_tool_call_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - last_message_with_tool_calls: dict | None, - forward_function_call_id: bool = False, -) -> VertexPartType | list[VertexPartType]: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "content": "function result goes here", - }, - - # NOTE: Function messages have been deprecated - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - - Supports content with images for Computer Use: - { - "role": "tool", - "tool_call_id": "call_abc123", - "content": [ - {"type": "text", "text": "I found the requested image:"}, - {"type": "input_image", "image_url": "https://example.com/image.jpg" } - ] - } - """ - from litellm.types.llms.vertex_ai import BlobType - - content_str: str = "" - inline_data_list: Final[list[BlobType]] = [] - - if "content" in message: - if isinstance(message["content"], str): - content_str = message["content"] - # Detect data-URL images (e.g. from Anthropic tool_result with a single image block - # that was serialised as a plain string by translate_anthropic_messages_to_openai) - # and promote them to inline_data so Gemini receives actual image bytes. - if content_str[:5].lower() == "data:" and ";base64," in content_str: - try: - mime_rest: Final = content_str[5:].split(";base64,", 1) - if len(mime_rest) == 2 and mime_rest[0].startswith("image/"): - # Strip any extra parameters (e.g. ";charset=UTF-8") from the MIME segment - clean_mime: Final = mime_rest[0].split(";")[0].strip() - inline_data_list.append(BlobType(data=mime_rest[1], mime_type=clean_mime)) - content_str = "" - except Exception as e: - verbose_logger.warning("Failed to parse data URL in tool response: %s", e) - elif isinstance(message["content"], list): - content_list: Final = message["content"] - for content in content_list: - content_type = content.get("type", "") - if content_type == "text": - content_str += content.get("text", "") - elif content_type == "image": # pyright: ignore[reportUnnecessaryComparison] # loose runtime dict - # Anthropic-native image block: {"type": "image", "source": {"type": "base64", ...}} - source = content.get("source", {}) - if isinstance(source, dict) and source.get("type") == "base64": - try: - inline_data_list.append( - BlobType( - data=source.get("data", ""), - mime_type=source.get("media_type", "image/jpeg"), - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process Anthropic image block in tool response: %s", e) - elif content_type in ("input_image", "image_url"): - # Extract image for inline_data (for Computer Use screenshots and tool results) - image_url_data = content.get("image_url", "") - image_url = image_url_data.get("url", "") if isinstance(image_url_data, dict) else image_url_data - - if image_url: - # Convert image to base64 blob format for Gemini - try: - image_obj = convert_to_anthropic_image_obj(image_url, format=None) - inline_data_list.append( - BlobType( - data=image_obj["data"], - mime_type=image_obj["media_type"], - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process image in tool response: %s", e) - elif content_type in ("file", "input_file"): # pyright: ignore[reportUnnecessaryContains] # loose runtime dict - # Extract file for inline_data (for tool results with PDF, audio, video, etc.) - file_data = content.get("file_data", "") - if not file_data: - file_content = content.get("file", {}) - file_data = ( - file_content.get("file_data", "") - if isinstance(file_content, dict) - else file_content - if isinstance(file_content, str) - else "" - ) - - if file_data: - # Convert file to base64 blob format for Gemini - try: - file_obj = convert_to_anthropic_image_obj(file_data, format=None) - inline_data_list.append( - BlobType( - data=file_obj["data"], - mime_type=file_obj["media_type"], - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process file in tool response: %s", e) - name: str | None = message.get("name", "") - - # Recover name from last message with tool calls - if last_message_with_tool_calls: - tools: Final = last_message_with_tool_calls.get("tool_calls", []) - msg_tool_call_id: Final = message.get("tool_call_id", None) - for tool in tools: - prev_tool_call_id = tool.get("id", None) - if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: - name = tool.get("function", {}).get("name", "") - - # Echo the OpenAI tool_call_id on functionResponse (strip thought-signature suffix). - gemini_call_id: str | None = None - if forward_function_call_id: - raw_tool_call_id: Final = message.get("tool_call_id") - if raw_tool_call_id and isinstance(raw_tool_call_id, str): - stripped_id: Final = raw_tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] - if stripped_id: - gemini_call_id = stripped_id - - if not name: - raise Exception( - f"Missing corresponding tool call for tool response message. Received - message={message}, last_message_with_tool_calls={last_message_with_tool_calls}" - ) - - # Parse response data - support both JSON string and plain string - # For Computer Use, the response should contain structured data like {"url": "..."} - response_data: dict - try: - if content_str.strip().startswith("{") or content_str.strip().startswith("["): - # Try to parse as JSON (for Computer Use structured responses) - parsed: Final = json.loads(content_str) - if isinstance(parsed, dict): - response_data = parsed # Use the parsed JSON directly - else: - response_data = {"content": content_str} - else: - response_data = {"content": content_str} - except (json.JSONDecodeError, ValueError): - # Not valid JSON, wrap in content field - response_data = {"content": content_str} - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - _function_response: Final = VertexFunctionResponse( - name=name, - response=response_data, - ) - if gemini_call_id: - _function_response["id"] = gemini_call_id - - _part: Final[VertexPartType] = {"function_response": _function_response} - - # For multimodal function responses, Gemini expects media parts nested - # inside functionResponse.parts instead of sibling content parts. - if inline_data_list: - _function_response["parts"] = [{"inline_data": inline_data} for inline_data in inline_data_list] - return [_part] - - return _part - - -_TOOL_USE_ID_FALLBACK: Final = "tool_use_id" -_ANTHROPIC_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_-]") -_BEDROCK_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_.:-]") -_BEDROCK_TOOL_USE_ID_MAX_LEN: Final = 64 -_BEDROCK_TOOL_USE_ID_HASH_LEN: Final = 8 - - -def _replace_invalid_tool_use_id_chars(tool_use_id: str, invalid_chars: re.Pattern[str]) -> str: - return invalid_chars.sub("_", tool_use_id) or _TOOL_USE_ID_FALLBACK - - -def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str: - """Anthropic requires tool_use_id to match ^[a-zA-Z0-9_-]+$.""" - return _replace_invalid_tool_use_id_chars(tool_use_id, _ANTHROPIC_TOOL_USE_ID_INVALID_CHARS) - - -def _sanitize_bedrock_tool_use_id(tool_use_id: str) -> str: - """ - Bedrock Converse requires toolUseId to match [a-zA-Z0-9_.:-]+ and be at most 64 chars. - Ids that need rewriting get a short hash of the original appended so two ids that only - differ in a replaced char or past the cut still map to distinct values. - """ - sanitized: Final = _replace_invalid_tool_use_id_chars(tool_use_id, _BEDROCK_TOOL_USE_ID_INVALID_CHARS) - if sanitized == tool_use_id and len(sanitized) <= _BEDROCK_TOOL_USE_ID_MAX_LEN: - return sanitized - digest: Final = hashlib.sha256(tool_use_id.encode()).hexdigest()[:_BEDROCK_TOOL_USE_ID_HASH_LEN] - return f"{sanitized[: _BEDROCK_TOOL_USE_ID_MAX_LEN - _BEDROCK_TOOL_USE_ID_HASH_LEN - 1]}_{digest}" - - -_ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES: Final = {"application/pdf", "text/plain"} - - -def _is_anthropic_document_data_uri(url: str) -> bool: - # Anthropic's base64 document source accepts only application/pdf and - # text/plain (see select_anthropic_content_block_type_for_file). Routing - # other mimes here would produce a document block the API rejects, so we - # leave them on the image code path. - match: Final = re.match(r"data:([^;,]+)", url) - if not match: - return False - return match.group(1) in _ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES - - -def convert_to_anthropic_tool_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - force_base64: bool = False, -) -> AnthropicMessagesToolResultParam: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - - """ - Anthropic tool_results look like: - { - "role": "user", - "content": [ - { - "type": "tool_result", - "tool_use_id": "toolu_01A09q90qw90lq917835lq9", - "content": "ConnectionError: the weather service API is not available (HTTP 500)", - # "is_error": true - } - ] - } - """ - anthropic_content: ( - str - | list[ - AnthropicMessagesToolResultContent - | AnthropicMessagesImageParam - | AnthropicMessagesDocumentParam - | ToolReference - ] - ) = "" - if isinstance(message["content"], str): - anthropic_content = message["content"] - elif isinstance(message["content"], list): - content_list: Final = message["content"] - anthropic_content_list: list[ - AnthropicMessagesToolResultContent - | AnthropicMessagesImageParam - | AnthropicMessagesDocumentParam - | ToolReference - ] = [] - for content in content_list: - if content["type"] == "text": - # Only include cache_control if explicitly set and not None - # to avoid sending "cache_control": null which breaks some API channels - text_content: AnthropicMessagesToolResultContent = { - "type": "text", - "text": content["text"], - } - cache_control_value = content.get("cache_control") - if cache_control_value is not None: - text_content["cache_control"] = cache_control_value - anthropic_content_list.append(text_content) - elif content["type"] == "image_url": - image_url_value = content["image_url"] - format = image_url_value.get("format") if isinstance(image_url_value, dict) else None - url_str = image_url_value.get("url") if isinstance(image_url_value, dict) else image_url_value - # Data URIs with non-image mime types (e.g. application/pdf) must - # translate to Anthropic document blocks, not image blocks — - # wrapping a PDF in `type: "image"` is rejected by the API. - if isinstance(url_str, str) and _is_anthropic_document_data_uri(url_str): - synth_file_message: ChatCompletionFileObject = { - "type": "file", - "file": {"file_data": url_str}, - } - _document_block = anthropic_process_openai_file_message(synth_file_message) - _document_block = add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, _document_block), - original_content_element=content, - ) - anthropic_content_list.append(cast(AnthropicMessagesDocumentParam, _document_block)) - else: - _anthropic_image_param = create_anthropic_image_param( - image_url_value, - format=format, - is_bedrock_invoke=force_base64, - ) - _anthropic_image_param = add_cache_control_to_content( - anthropic_content_element=_anthropic_image_param, - original_content_element=content, - ) - anthropic_content_list.append(cast(AnthropicMessagesImageParam, _anthropic_image_param)) - elif content["type"] == "tool_reference": - anthropic_content_list.append(ToolReference(type="tool_reference", tool_name=content["tool_name"])) - elif content["type"] == "file": - file_content = cast(ChatCompletionFileObject, content) - _file_block = anthropic_process_openai_file_message(file_content) - _file_block = add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, _file_block), - original_content_element=content, - ) - anthropic_content_list.append(_file_block) - - anthropic_content = anthropic_content_list - anthropic_tool_result: AnthropicMessagesToolResultParam | None = None - ## PROMPT CACHING CHECK ## - cache_control: Final = message.get("cache_control", None) - if message["role"] == "tool": - tool_message: Final[ChatCompletionToolMessage] = message - tool_call_id: str = tool_message["tool_call_id"] - # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ - sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", - tool_use_id=sanitized_tool_use_id, - content=anthropic_content, - ) - - if message["role"] == "function": - function_message: Final[ChatCompletionFunctionMessage] = message - tool_call_id = function_message.get("tool_call_id") or str(uuid.uuid4()) - # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ - sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) - anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", - tool_use_id=sanitized_tool_use_id, - content=anthropic_content, - ) - - if anthropic_tool_result is None: - raise Exception(f"Unable to parse anthropic tool result for message: {message}") - if cache_control is not None: - anthropic_tool_result["cache_control"] = cache_control - return anthropic_tool_result - - -def convert_function_to_anthropic_tool_invoke( - function_call: dict | ChatCompletionToolCallFunctionChunk, -) -> list[AnthropicMessagesToolUseParam]: - try: - _name: Final = get_attribute_or_key(function_call, "name") or "" - _arguments: Final = get_attribute_or_key(function_call, "arguments") - - tool_input: Final = parse_tool_call_arguments( - _arguments, tool_name=_name, context="Anthropic function to tool invoke" - ) - - anthropic_tool_invoke: Final = [ - AnthropicMessagesToolUseParam( - type="tool_use", - id=str(uuid.uuid4()), - name=_name, - input=tool_input, - ) - ] - return anthropic_tool_invoke - except Exception as e: - raise e - - -def _find_server_tool_result( - tool_id: str, - web_search_results: Sequence[object] | None, - tool_results: Sequence[object] | None, -) -> dict[str, object] | None: - candidates: Final = (*(web_search_results or ()), *(tool_results or ())) - return next( - (result for result in candidates if isinstance(result, dict) and result.get("tool_use_id") == tool_id), - None, - ) - - -def convert_to_anthropic_tool_invoke( - tool_calls: list[ChatCompletionAssistantToolCall], - web_search_results: Sequence[object] | None = None, - tool_results: Sequence[object] | None = None, -) -> list[AnthropicMessagesToolUseParam | dict[str, Any]]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - - """ - Anthropic tool invokes: - { - "role": "assistant", - "content": [ - { - "type": "text", - "text": "To answer this question, I will: 1. Use the get_weather tool to get the current weather in San Francisco. 2. Use the get_time tool to get the current time in the America/Los_Angeles timezone, which covers San Francisco, CA." - }, - { - "type": "tool_use", - "id": "toolu_01A09q90qw90lq917835lq9", - "name": "get_weather", - "input": {"location": "San Francisco, CA"} - } - ] - } - - For server-side tools (web_search), we need to reconstruct: - - server_tool_use blocks (id starts with "srvtoolu_") - - web_search_tool_result blocks (from provider_specific_fields) - - Fixes: https://github.com/BerriAI/litellm/issues/17737 - """ - anthropic_tool_invoke: Final[list[AnthropicMessagesToolUseParam | dict[str, object]]] = [] - - for tool in tool_calls: - if get_attribute_or_key(tool, "type") != "function": - continue - - tool_id = cast(str, get_attribute_or_key(tool, "id")) - tool_name = cast( - str, - get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), - ) - tool_input = parse_tool_call_arguments( - get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments"), - tool_name=tool_name, - context="Anthropic tool invoke", - ) - - server_tool_result = ( - _find_server_tool_result(tool_id, web_search_results, tool_results) - if tool_id.startswith("srvtoolu_") - else None - ) - if server_tool_result is not None: - anthropic_tool_invoke.append( - { - "type": "server_tool_use", - "id": tool_id, - "name": tool_name, - "input": tool_input, - } - ) - anthropic_tool_invoke.append(server_tool_result) - else: - sanitized_tool_id = _sanitize_anthropic_tool_use_id(tool_id) - _anthropic_tool_use_param = AnthropicMessagesToolUseParam( - type="tool_use", - id=sanitized_tool_id, - name=tool_name, - input=tool_input, - ) - - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_tool_use_param, - original_content_element=dict(tool), - ) - - if "cache_control" in _content_element: - _anthropic_tool_use_param["cache_control"] = _content_element["cache_control"] - - anthropic_tool_invoke.append(_anthropic_tool_use_param) - - return anthropic_tool_invoke - - -def add_cache_control_to_content( - anthropic_content_element: dict - | AnthropicMessagesImageParam - | AnthropicMessagesTextParam - | AnthropicMessagesDocumentParam - | AnthropicMessagesToolUseParam - | ChatCompletionThinkingBlock, - original_content_element: dict | AllMessageValues, -): - cache_control_param: Final = original_content_element.get("cache_control") - if cache_control_param is not None and isinstance(cache_control_param, dict): - transformed_param: Final = ChatCompletionCachedContent(**cache_control_param) - - anthropic_content_element["cache_control"] = transformed_param - - return anthropic_content_element - - -def _anthropic_content_element_factory( - image_chunk: GenericImageParsingChunk, -) -> AnthropicMessagesImageParam | AnthropicMessagesDocumentParam: - if image_chunk["media_type"] == "application/pdf": - _anthropic_content_element: AnthropicMessagesDocumentParam | AnthropicMessagesImageParam = ( - AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - ) - else: - _anthropic_content_element = AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - - return _anthropic_content_element - - -def select_anthropic_content_block_type_for_file( - format: str, -) -> Literal["document", "image", "container_upload"]: - if format == "application/pdf" or format == "text/plain": - return "document" - elif format in ["image/jpeg", "image/png", "image/gif", "image/webp"]: - return "image" - else: - return "container_upload" - - -def anthropic_infer_file_id_content_type( - file_id: str, -) -> Literal["document_url", "container_upload"]: - """ - Use when 'format' not provided. - - - URL's - assume are document_url - - Else - assume is container_upload - """ - if file_id.startswith("http") or file_id.startswith("https"): - return "document_url" - else: - return "container_upload" - - -def anthropic_process_openai_file_message( - message: ChatCompletionFileObject, -) -> AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam: - file_message: Final = cast(ChatCompletionFileObject, message) - file_sub: Final = file_message.get("file") - if file_sub is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="anthropic", - ) - file_data: Final = file_sub.get("file_data") - file_id: Final = file_sub.get("file_id") - format: Final = file_sub.get("format") - if file_data: - image_chunk: Final = convert_to_anthropic_image_obj( - openai_image_url=file_data, - format=format, - ) - anthropic_document_param: Final = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - return anthropic_document_param - elif file_id: - content_block_type: Final = ( - select_anthropic_content_block_type_for_file(format) - if format - else anthropic_infer_file_id_content_type(file_id) - ) - return_block_param: ( - AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam | None - ) = None - if content_block_type == "document": - return_block_param = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSourceFileId( - type="file", - file_id=file_id, - ), - ) - elif content_block_type == "document_url": - return_block_param = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSourceUrl( - type="url", - url=file_id, - ), - ) - elif content_block_type == "image": - return_block_param = AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSourceFileId( - type="file", - file_id=file_id, - ), - ) - elif content_block_type == "container_upload": - return_block_param = AnthropicMessagesContainerUploadParam(type="container_upload", file_id=file_id) - - if return_block_param is None: - raise Exception(f"Unable to parse anthropic file message: {message}") - return return_block_param - raise Exception(f"Either file_data or file_id must be present in the file message: {message}") - - -_EMPTY_TEXT_PLACEHOLDER: Final = "[System: Empty message content sanitised to satisfy protocol]" - - -def _sanitize_empty_text_content( - message: AllMessageValues, -) -> AllMessageValues: - """ - Case C: Sanitize empty text content - - Replace empty or whitespace-only text content with a placeholder message. - - Handles both string content and list-of-blocks content (rewriting only - the empty text blocks in place; non-text blocks like images are left - untouched). - - Returns: - The message with sanitized content if needed, otherwise the original message - """ - if message.get("role") not in ["user", "assistant"]: - return message - - content: Final = message.get("content") - - if isinstance(content, str): - if not content or not content.strip(): - message = cast(AllMessageValues, dict(message)) # Make a copy - message["content"] = _EMPTY_TEXT_PLACEHOLDER - verbose_logger.debug( - "_sanitize_empty_text_content: Replaced empty text content in %s message", message.get("role") - ) - return message - - if isinstance(content, list): - # Walk the blocks and rewrite any empty text blocks. We rewrite (rather - # than drop) so callers don't end up with an entirely empty content - # list, which Anthropic also rejects. - new_blocks: Final[list[Any]] = [] - rewrote_any = False - for block in content: - if isinstance(block, dict) and block.get("type") == "text": - text = block.get("text") - if not isinstance(text, str) or not text or not text.strip(): - new_block = dict(block) - new_block["text"] = _EMPTY_TEXT_PLACEHOLDER - new_blocks.append(new_block) - rewrote_any = True - continue - new_blocks.append(block) - - if rewrote_any: - message = cast(AllMessageValues, dict(message)) # Make a copy - message["content"] = new_blocks - verbose_logger.debug( - "_sanitize_empty_text_content: Replaced empty text block(s) in %s message", message.get("role") - ) - - return message - - -def _add_missing_tool_results( - current_message: AllMessageValues, - messages: list[AllMessageValues], - current_index: int, -) -> tuple[list[AllMessageValues], int]: - """ - Case A: Missing tool_result for tool_use (orphaned tool calls) - - If an assistant message has tool_calls but no corresponding tool result follows, - add a dummy tool result message indicating the user did not provide the result. - - Returns: - A tuple of: - - List containing the assistant message, followed by existing tool results, - followed by any dummy tool results needed - - Number of original messages consumed (to adjust iteration index) - """ - result_messages: Final[list[AllMessageValues]] = [] - tool_calls: Final = current_message.get("tool_calls") - - if not tool_calls or len(cast(list, tool_calls)) == 0: - return ([current_message], 0) - - # Collect all tool_call_ids from this assistant message - expected_tool_call_ids: Final = set() - for tool_call in cast(list, tool_calls): - tool_call_id = None - if isinstance(tool_call, dict): - tool_call_id = tool_call.get("id") - else: - tool_call_id = getattr(tool_call, "id", None) - if tool_call_id: - expected_tool_call_ids.add(tool_call_id) - - # Collect actual tool result messages that follow this assistant message - found_tool_call_ids: Final = set() - actual_tool_results: Final[list[AllMessageValues]] = [] - j = current_index + 1 - - while j < len(messages): - next_msg = messages[j] - next_role = next_msg.get("role") - - if next_role == "assistant": - break - - if next_role in ["tool", "function"]: - tool_call_id = next_msg.get("tool_call_id") - if tool_call_id and tool_call_id in expected_tool_call_ids: - found_tool_call_ids.add(tool_call_id) - actual_tool_results.append(next_msg) - - j += 1 - - # Find missing tool results - missing_tool_call_ids: Final = expected_tool_call_ids - found_tool_call_ids - - if missing_tool_call_ids: - verbose_logger.debug( - "_add_missing_tool_results: Found %s orphaned tool calls. Adding dummy tool results.", - len(missing_tool_call_ids), - ) - - result_messages.append(current_message) - - # Add existing tool results FIRST - result_messages.extend(actual_tool_results) - - # Then add dummy tool results for missing ones - for tool_call_id in missing_tool_call_ids: - tool_name = "unknown_tool" - for tool_call in cast(list, tool_calls): - tc_id = None - if isinstance(tool_call, dict): - tc_id = tool_call.get("id") - else: - tc_id = getattr(tool_call, "id", None) - - if tc_id == tool_call_id: - if isinstance(tool_call, dict): - function = tool_call.get("function", {}) - if isinstance(function, dict): - tool_name = function.get("name", "unknown_tool") - else: - tool_name = getattr(function, "name", "unknown_tool") - else: - function = getattr(tool_call, "function", None) - if function: - tool_name = getattr(function, "name", "unknown_tool") - break - - dummy_tool_result: ChatCompletionToolMessage = { - "role": "tool", - "tool_call_id": tool_call_id, - "content": f"[System: Tool execution skipped/interrupted by user. No result provided for tool '{tool_name}'.]", - } - result_messages.append(dummy_tool_result) - - # Return the messages and the number of original messages to skip - return (result_messages, len(actual_tool_results)) - - return ([current_message], 0) - - -def _is_orphaned_tool_result( - current_message: AllMessageValues, - sanitized_messages: list[AllMessageValues], -) -> bool: - """ - Case B: Orphaned tool_result (unexpected result) - - Check if a tool message references a tool_call_id that doesn't exist in the previous - assistant message. - - Returns: - True if this is an orphaned tool result that should be removed, False otherwise - """ - if current_message.get("role") not in ["tool", "function"]: - return False - - tool_call_id: Final = current_message.get("tool_call_id") - - if not tool_call_id: - return False - - # Look back to find the most recent assistant message with tool_calls - found_matching_tool_call = False - - for j in range(len(sanitized_messages) - 1, -1, -1): - prev_msg = sanitized_messages[j] - if prev_msg.get("role") == "assistant": - tool_calls = prev_msg.get("tool_calls") - if tool_calls: - for tool_call in cast(list, tool_calls): - tc_id = None - if isinstance(tool_call, dict): - tc_id = tool_call.get("id") - else: - tc_id = getattr(tool_call, "id", None) - - if tc_id == tool_call_id: - found_matching_tool_call = True - break - - break - - if not found_matching_tool_call: - verbose_logger.debug("_is_orphaned_tool_result: Found orphaned tool result with redacted tool_call_id") - return True - - return False - - -def _declared_tool_call_ids(message: Mapping[str, object]) -> frozenset[str]: - tool_calls: Final = message.get("tool_calls") - if not isinstance(tool_calls, list): - return frozenset() - return frozenset( - str(tool_call["id"]) for tool_call in tool_calls if isinstance(tool_call, Mapping) and tool_call.get("id") - ) - - -def group_tool_exchanges(messages: Sequence[Mapping[str, object]]) -> tuple[tuple[int, ...], ...]: - """Group message indices into tool exchanges: an assistant row that made - tool calls, together with the tool rows answering the ids it declared. - - Membership is by ``tool_call_id`` ownership rather than adjacency, so a tool - row belonging to some other call opens its own group instead of being swept - into the exchange it happens to sit next to. Every other row is its own - group. Groups stay contiguous and in order, so a caller can convert or - protect them without reordering the conversation. - - Callers need this because an assistant row and the tool rows answering it - are only well-formed together: ``sanitize_messages_for_tool_calling`` reads - an assistant row whose results are missing as an orphaned tool call, and - a tool row whose call is missing as an orphaned result. - """ - return tuple(_iter_tool_exchange_groups(messages)) - - -def _iter_tool_exchange_groups(messages: Sequence[Mapping[str, object]]) -> Iterator[tuple[int, ...]]: - index = 0 - while index < len(messages): - declared = _declared_tool_call_ids(messages[index]) - end = index + 1 - while ( - declared - and end < len(messages) - and messages[end].get("role") in ("tool", "function") - and str(messages[end].get("tool_call_id")) in declared - ): - end += 1 - yield tuple(range(index, end)) - index = end - - -def sanitize_messages_for_tool_calling( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: - """ - Sanitize messages for tool calling to handle common issues when modify_params=True: - - Case A: Missing tool_result for tool_use (orphaned tool calls) - - If an assistant message has tool_calls but no corresponding tool result follows, - add a dummy tool result message indicating the user did not provide the result. - - Case B: Orphaned tool_result (unexpected result) - - If a tool message references a tool_call_id that doesn't exist in the previous - assistant message, remove that tool message. - - Case C: Empty text content - - Replace empty or whitespace-only text content with a placeholder message. - - Case D: Duplicate tool_result for same tool_use (duplicate results) - - If multiple tool messages reference the same tool_call_id, keep only the last - occurrence. Anthropic requires exactly one tool_result per tool_use and rejects - with: "each tool_use must have a single result". - - This function operates on OpenAI format messages before they are converted to - provider-specific formats. - """ - if not litellm.modify_params: - return messages - - sanitized_messages: list[AllMessageValues] = [] - i = 0 - - while i < len(messages): - current_message = messages[i] - - # Case C: Sanitize empty text content - current_message = _sanitize_empty_text_content(current_message) - - # Case A: Check if assistant message has tool_calls without following tool results - if current_message.get("role") == "assistant": - result_messages, messages_consumed = _add_missing_tool_results(current_message, messages, i) - - # If dummy tool results were added, extend sanitized_messages and skip consumed messages - if len(result_messages) > 1: - sanitized_messages.extend(result_messages) - # Skip the assistant message and any actual tool results that were included - i += 1 + messages_consumed - continue - - # Case B: Check for orphaned tool results - if _is_orphaned_tool_result(current_message, sanitized_messages): - i += 1 - continue # Skip this orphaned tool result - - # Add the message to sanitized list - sanitized_messages.append(current_message) - i += 1 - - # Case D: Deduplicate tool results with the same tool_call_id. - # Anthropic requires exactly one tool_result per tool_use. Session history - # (e.g. from conversation resume) can contain duplicate tool_result messages - # for the same tool_call_id. Keep only the last occurrence *within each - # contiguous block of tool results following an assistant message*. This - # avoids dropping results from earlier turns if a tool_call_id is reused. - # - # NOTE: This intentionally keeps the *last* occurrence (most complete for - # session-resume duplicates), unlike _deduplicate_bedrock_content_blocks - # which keeps the *first*. The Bedrock case handles provider-side content - # block duplication where the first is authoritative; here the duplicate - # arises from history replay where the last entry is the final state. - duplicates_to_remove: Final[set[int]] = set() - seen_in_block: dict[str, int] = {} # tool_call_id -> index (reset per block) - for idx, msg in enumerate(sanitized_messages): - role = msg.get("role") - tcid = msg.get("tool_call_id") if role in ["tool", "function"] else None - if tcid and isinstance(tcid, str): - if tcid in seen_in_block: - # Mark the earlier occurrence for removal (keep latest) - duplicates_to_remove.add(seen_in_block[tcid]) - verbose_logger.warning( - "sanitize_messages_for_tool_calling: dropping duplicate " - "tool_result with tool_call_id=%s. This may indicate " - "duplicate tool messages in conversation history.", - tcid, - ) - seen_in_block[tcid] = idx - elif role not in ("tool", "function"): - # Non-tool message (user, assistant, system) marks a - # conversational-turn boundary — reset tracking. - # Tool/function messages with no tool_call_id are malformed; - # they should NOT reset the block because they don't represent - # a turn boundary and would mask real within-block duplicates. - seen_in_block = {} - - if duplicates_to_remove: - sanitized_messages = [msg for idx, msg in enumerate(sanitized_messages) if idx not in duplicates_to_remove] - - return sanitized_messages - - -def _drop_unsignable_thinking_blocks( - thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], -) -> list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock]: - return [block for block in thinking_blocks if not is_unsignable_thinking_block(block)] - - -_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues] - - -def anthropic_messages_pt( - messages: list[AllMessageValues], - model: str, - llm_provider: str, -) -> _AnthropicMessageList: - """ - format messages for anthropic - 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately). - Models flagged ``supports_mid_conversation_system`` also accept "system" inside - messages after a user turn; the caller decides placement, this keeps such messages. - 2. The first message always needs to be of role "user" - 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) - 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) - 5. System messages are a separate param to the Messages API - 6. Ensure we only accept role, content. (message.name is not supported) - """ - # Sanitize messages for tool calling issues when modify_params=True - messages = sanitize_messages_for_tool_calling(messages) - - # Anthropic rejects empty text content blocks with: - # "messages: text content blocks must be non-empty" - # OpenAI/other providers silently tolerate `{"role": "user", "content": ""}`, - # so callers (and upstream agent frameworks like pydantic-ai) routinely - # send empty user/assistant turns. We always rewrite these to a placeholder - # for Anthropic-shaped requests, independent of `litellm.modify_params`, - # because there is no way to "pass through" an empty text block — the - # request will always 400 otherwise. The richer tool-call sanitization - # (Cases A/B/D in `sanitize_messages_for_tool_calling`) remains gated on - # `modify_params` because it actually mutates conversation structure. - messages = [_sanitize_empty_text_content(m) for m in messages] - - # add role=tool support to allow function call result/error submission - user_message_types: Final = {"user", "tool", "function"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract - - if len(messages) == 0: - if not litellm.modify_params: - raise litellm.BadRequestError( - message=f"Anthropic requires at least one non-system message. Either provide one, or set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add the dummy user message - {DEFAULT_USER_CONTINUE_MESSAGE_TYPED}.", - model=model, - llm_provider=llm_provider, - ) - else: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE_TYPED) - - # Bedrock invoke models have format: invoke/... - # Vertex AI Anthropic also doesn't support URL sources for images - is_bedrock_invoke = model.lower().startswith("invoke/") - is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False - force_base64 = is_bedrock_invoke or is_vertex_ai - - msg_i = 0 - while msg_i < len(messages): - user_content: list[AnthropicMessagesUserMessageValues] = [] - init_msg_i = msg_i - if isinstance(messages[msg_i], BaseModel): - messages[msg_i] = dict(messages[msg_i]) - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - user_message_types_block: ( - ChatCompletionToolMessage | ChatCompletionUserMessage | ChatCompletionFunctionMessage - ) = messages[msg_i] - if user_message_types_block["role"] == "user": - if isinstance(user_message_types_block["content"], list): - user_content_start: int = len(user_content) - for m in user_message_types_block["content"]: - if m.get("type", "") == "image_url": - m = cast(ChatCompletionImageObject, m) - format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None - # Convert ChatCompletionImageUrlObject to dict if needed - image_url_value = m["image_url"] - if isinstance(image_url_value, str): - image_url_input: str | dict[str, object] = image_url_value - else: - # ChatCompletionImageUrlObject or dict case - convert to dict - image_url_input = { - "url": image_url_value["url"], - "format": image_url_value.get("format"), - } - # Bedrock invoke models have format: invoke/... - # Vertex AI Anthropic also doesn't support URL sources for images - is_bedrock_invoke = model.lower().startswith("invoke/") - is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False - force_base64 = is_bedrock_invoke or is_vertex_ai - _anthropic_content_element = create_anthropic_image_param( - image_url_input, - format=format, - is_bedrock_invoke=force_base64, - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_content_element, - original_content_element=dict(m), - ) - - if "cache_control" in _content_element: - _anthropic_content_element["cache_control"] = _content_element["cache_control"] - user_content.append(_anthropic_content_element) - elif m.get("type", "") == "text": - m = cast(ChatCompletionTextObject, m) - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=m["text"], - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(m), - ) - _content_element = cast(AnthropicMessagesTextParam, _content_element) - - user_content.append(_content_element) - elif m.get("type", "") == "document": - _document_content_element = cast( - AnthropicMessagesDocumentParam, - add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, m), - original_content_element=dict(m), - ), - ) - user_content.append(_document_content_element) - elif m.get("type", "") == "file": - _file_content_element = anthropic_process_openai_file_message( - cast(ChatCompletionFileObject, m) - ) - _file_content_element = add_cache_control_to_content( - anthropic_content_element=cast( - AnthropicMessagesDocumentParam, - _file_content_element, - ), - original_content_element=dict(m), - ) - user_content.append( - cast( - AnthropicMessagesDocumentParam, - _file_content_element, - ) - ) - if ( - len(user_content) > user_content_start - and user_message_types_block.get("cache_control") is not None - and not any( - "cache_control" in content and content["cache_control"] is not None - for content in user_content[user_content_start:] - ) - ): - cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( - dict[str, object], cached_user_content - ), # cast-ok: copied TypedDict is a dict - original_content_element=user_message_types_block, - ) - user_content[-1] = cached_user_content - elif isinstance(user_message_types_block["content"], str): - _anthropic_content_text_element: AnthropicMessagesTextParam = { - "type": "text", - "text": user_message_types_block["content"], - } - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_content_text_element, - original_content_element=dict(user_message_types_block), - ) - - if "cache_control" in _content_element: - _anthropic_content_text_element["cache_control"] = _content_element["cache_control"] - - user_content.append(_anthropic_content_text_element) - - elif user_message_types_block["role"] == "tool" or user_message_types_block["role"] == "function": - # OpenAI's tool message content will always be a string - user_content.append( - convert_to_anthropic_tool_result(user_message_types_block, force_base64=force_base64) - ) - - msg_i += 1 - - if user_content: - new_messages.append({"role": "user", "content": user_content}) - - # Track unique tool IDs in this merge block to avoid duplication - unique_tool_ids: set[str] = set() - - assistant_content: list[AnthropicMessagesAssistantMessageValues] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_content_block: ChatCompletionAssistantMessage = messages[msg_i] - - # Extract compaction_blocks from provider_specific_fields and add them first - _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") - if isinstance(_provider_specific_fields_raw, dict): - _compaction_blocks = _provider_specific_fields_raw.get("compaction_blocks") - if _compaction_blocks and isinstance(_compaction_blocks, list): - # Add compaction blocks at the beginning of assistant content : https://platform.claude.com/docs/en/build-with-claude/compaction - assistant_content.extend(_compaction_blocks) - - _raw_thinking_blocks = assistant_content_block.get("thinking_blocks", None) - thinking_blocks = ( - _drop_unsignable_thinking_blocks(_raw_thinking_blocks) if _raw_thinking_blocks is not None else None - ) - - # Check if tool_calls contain server tool calls (web search, etc.) - # If so, we need to interleave thinking blocks with tool call groups - # to preserve the original content block ordering. - # Fixes: https://github.com/BerriAI/litellm/issues/23047 - assistant_tool_calls = assistant_content_block.get("tool_calls") - _has_server_tool_calls = False - if assistant_tool_calls is not None: - for _tc in assistant_tool_calls: - _tc_id = _tc.get("id") if isinstance(_tc, dict) else getattr(_tc, "id", None) - if _tc_id and isinstance(_tc_id, str) and _tc_id.startswith("srvtoolu_"): - _has_server_tool_calls = True - break - - _content_is_list: bool = "content" in assistant_content_block and isinstance( - assistant_content_block["content"], list - ) - _content_list = assistant_content_block.get("content") if _content_is_list else None - assistant_content_start: int = len(assistant_content) - - if ( - thinking_blocks is not None - and _has_server_tool_calls - and isinstance(assistant_content_block.get("content", None), (str, type(None))) - ): - # INTERLEAVED MODE: When we have both thinking blocks and server - # tool calls (e.g. web search), Anthropic's original response - # interleaves them: [thinking_1, server_tool_use_1, result_1, - # thinking_2, text, server_tool_use_2, result_2, ...]. - # We must preserve this interleaved order because Anthropic - # verifies thinking block signatures based on position. - - # Build the tool call groups (server_tool_use + its result) - _provider_specific_fields_raw_tc = assistant_content_block.get("provider_specific_fields") - _provider_specific_fields_tc: dict[str, Any] = {} - if isinstance(_provider_specific_fields_raw_tc, dict): - _provider_specific_fields_tc = cast(dict[str, Any], _provider_specific_fields_raw_tc) - _web_search_results_tc = _provider_specific_fields_tc.get("web_search_results") - _tool_results_tc = _provider_specific_fields_tc.get("tool_results") - tool_invoke_results = convert_to_anthropic_tool_invoke( - assistant_tool_calls, - web_search_results=_web_search_results_tc, - tool_results=_tool_results_tc, - ) - - # Group tool invoke results into (server_tool_use, result) pairs - # and separate regular tool_use blocks - server_tool_groups: list[list[Any]] = [] - regular_tool_uses: list[Any] = [] - _current_group: list[Any] = [] - for item in tool_invoke_results: - item_type = item.get("type", "") if isinstance(item, dict) else getattr(item, "type", "") - if item_type == "server_tool_use": - if _current_group: - server_tool_groups.append(_current_group) - _current_group = [item] - elif item_type.endswith("_tool_result"): - _current_group.append(item) - elif item_type == "tool_use": - regular_tool_uses.append(item) - else: - _current_group.append(item) - if _current_group: - server_tool_groups.append(_current_group) - - # Build the text block if content is a non-empty string - text_element = None - _acb_content = assistant_content_block.get("content") - if isinstance(_acb_content, str) and _acb_content: - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=_acb_content, - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(assistant_content_block), - ) - if "cache_control" in _content_element: - _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] - text_element = _anthropic_text_content_element - - # Interleave: each thinking block precedes its server tool group. - # Pattern: thinking[0], group[0], thinking[1], group[1], ... - # Any remaining thinking blocks (after all groups) go before text. - # Any remaining groups (after all thinking blocks) go after. - tb_idx = 0 - grp_idx = 0 - num_tb = len(thinking_blocks) if thinking_blocks else 0 - num_grp = len(server_tool_groups) - - while tb_idx < num_tb or grp_idx < num_grp: - if tb_idx < num_tb and grp_idx < num_grp: - # Emit thinking block then its tool group - assistant_content.append(thinking_blocks[tb_idx]) - tb_idx += 1 - for block in server_tool_groups[grp_idx]: - item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) - grp_idx += 1 - elif tb_idx < num_tb: - # More thinking blocks than tool groups - emit before text - assistant_content.append(thinking_blocks[tb_idx]) - tb_idx += 1 - else: - # More tool groups than thinking blocks - emit remaining - for block in server_tool_groups[grp_idx]: - item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) - grp_idx += 1 - - # Add text block (if any) - if text_element is not None: - assistant_content.append(text_element) - - # Add regular (non-server) tool calls at the end - for item in regular_tool_uses: - item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) - - # Mark tool_calls as already processed so they are not added again - assistant_tool_calls = None - - else: - # SEQUENTIAL MODE: No server tool calls, or no thinking blocks, - # or content is a list. Use the original sequential approach. - - # When content is a list, check if it already contains thinking - # blocks inline. If so, skip prepending thinking_blocks to avoid - # duplication and preserve the original interleaved order. - # Fixes the gap where list-content messages bypass INTERLEAVED - # MODE and still get thinking blocks prepended out of order. - _list_has_thinking = False - if _content_is_list and _content_list is not None: - for _item in _content_list: - if isinstance(_item, dict) and _item.get("type") in ( - "thinking", - "redacted_thinking", - ): - _list_has_thinking = True - break - - if ( - thinking_blocks is not None and not _list_has_thinking - ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR - assistant_content.extend(thinking_blocks) - if _content_is_list and _content_list is not None: - assistant_content_start = len(assistant_content) - for m in _content_list: - if not isinstance(m, dict): - continue - # handle thinking blocks - thinking_block = cast(str, m.get("thinking", "")) - text_block = cast(str, m.get("text", "")) - if ( - m.get("type", "") == "thinking" - and len(thinking_block) > 0 - and not is_unsignable_thinking_block(m) - ): # don't pass empty text blocks. anthropic api raises errors. - anthropic_message: ChatCompletionThinkingBlock | AnthropicMessagesTextParam = cast( - ChatCompletionThinkingBlock, m - ) - assistant_content.append(anthropic_message) - # handle text - elif ( - m.get("type", "") == "text" and len(text_block) > 0 - ): # don't pass empty text blocks. anthropic api raises errors. - anthropic_message = AnthropicMessagesTextParam(type="text", text=text_block) - _cached_message = add_cache_control_to_content( - anthropic_content_element=anthropic_message, - original_content_element=dict(m), - ) - - assistant_content.append(cast(AnthropicMessagesTextParam, _cached_message)) - # handle server_tool_use blocks (tool search, web search, etc.) - # Pass through as-is since these are Anthropic-native content types - elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): - assistant_content.append(m) - elif ( - "content" in assistant_content_block - and isinstance(assistant_content_block["content"], str) - and assistant_content_block["content"] # don't pass empty text blocks. anthropic api raises errors. - ): - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=assistant_content_block["content"], - ) - - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(assistant_content_block), - ) - - if "cache_control" in _content_element: - _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] - - assistant_content.append(_anthropic_text_content_element) - - if assistant_tool_calls is not None: # support assistant tool invoke conversion - # Get web_search_results and tool_results from provider_specific_fields - # for server_tool_use reconstruction. - # Fixes: https://github.com/BerriAI/litellm/issues/17737 - _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") - _provider_specific_fields: dict[str, Any] = {} - if isinstance(_provider_specific_fields_raw, dict): - _provider_specific_fields = cast(dict[str, Any], _provider_specific_fields_raw) - _web_search_results = _provider_specific_fields.get("web_search_results") - _tool_results = _provider_specific_fields.get("tool_results") - tool_invoke_results = convert_to_anthropic_tool_invoke( - assistant_tool_calls, - web_search_results=_web_search_results, - tool_results=_tool_results, - ) - - # Prevent "tool_use ids must be unique" errors by filtering duplicates - # This can happen when merging history that already contains the tool calls - for item in tool_invoke_results: - # tool_use items are typically dicts, but handle objects just in case - item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) - - if item_id: - if item_id in unique_tool_ids: - continue - unique_tool_ids.add(item_id) - - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) - - assistant_function_call = assistant_content_block.get("function_call") - - if assistant_function_call is not None: - assistant_content.extend(convert_function_to_anthropic_tool_invoke(assistant_function_call)) - - if ( - _content_is_list - and len(assistant_content) > assistant_content_start - and assistant_content_block.get("cache_control") is not None - and not any( - "cache_control" in content and content["cache_control"] is not None - for content in assistant_content[assistant_content_start:] - ) - ): - cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( - dict[str, object], cached_assistant_content - ), # cast-ok: copied TypedDict is a dict - original_content_element=assistant_content_block, - ) - assistant_content[-1] = cached_assistant_content - - msg_i += 1 - - if assistant_content: - new_messages.append({"role": "assistant", "content": assistant_content}) - - ## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - new_messages.extend(anthropic_system_messages(messages[msg_i])) - msg_i += 1 - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - if len(new_messages) > 0 and new_messages[-1]["role"] == "assistant": - if isinstance(new_messages[-1]["content"], str): - new_messages[-1]["content"] = new_messages[-1]["content"].rstrip() - elif isinstance(new_messages[-1]["content"], list): - for content in new_messages[-1]["content"]: - if isinstance(content, dict) and content["type"] == "text": - content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message - - return new_messages - - -def extract_between_tags(tag: str, string: str, strip: bool = False) -> list[str]: - ext_list = re.findall(f"<{tag}>(.+?)", string, re.DOTALL) - if strip: - ext_list = [e.strip() for e in ext_list] - return ext_list - - -def contains_tag(tag: str, string: str) -> bool: - return bool(re.search(f"<{tag}>(.+?)", string, re.DOTALL)) - - -def parse_xml_params(xml_content, json_schema: dict | None = None): - """ - Compare the xml output to the json schema - - check if a value is a list - if so, get it's child elements - """ - root: Final = ET.fromstring(xml_content) - params: Final = {} - - if json_schema is not None: # check if we have a json schema for this function call - # iterate over all properties in the schema - for prop in json_schema["properties"]: - # If property is an array, get the nested items - _element = root.find(f"parameters/{prop}") - if json_schema["properties"][prop]["type"] == "array": - items = [] - if _element is not None: - for value in _element: - try: - if value.text is not None: - _value = json.loads(value.text) - else: - continue - except json.JSONDecodeError: - _value = value.text - items.append(_value) - params[prop] = items - # If property is not an array, append the value directly - elif _element is not None and _element.text is not None: - try: - _value = json.loads(_element.text) - except json.JSONDecodeError: - _value = _element.text - params[prop] = _value - else: - for child in root.findall(".//parameters/*"): - if child is not None and child.text is not None: - try: - # Attempt to decode the element's text as JSON - params[child.tag] = json.loads(child.text) - except json.JSONDecodeError: - # If JSON decoding fails, use the original text - params[child.tag] = child.text - - return params - - -### GEMINI HELPER FUNCTIONS ### - - -def get_system_prompt(messages): - system_prompt_indices: Final = [] - system_prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "system": - system_prompt += message["content"] - system_prompt_indices.append(idx) - if len(system_prompt_indices) > 0: - for idx in reversed(system_prompt_indices): - messages.pop(idx) - return system_prompt, messages - - -from litellm.types.llms.cohere import ( - CallObject, - ChatHistory, - ChatHistoryChatBot, - ChatHistorySystem, - ChatHistoryToolResult, - ChatHistoryUser, - ToolCallObject, - ToolResultObject, -) - - -def convert_openai_message_to_cohere_tool_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - tool_calls: list, -) -> ToolResultObject: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "content": {"location": "San Francisco, CA", "unit": "fahrenheit", "temperature": "72"}, - }, - """ - """ - OpenAI message with a function call looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - - """ - Cohere tool_results look like: - { - "call": { - "name": "query_daily_sales_report", - "parameters": { - "day": "2023-09-29" - }, - }, - "outputs": [ - { - "date": "2023-09-29", - "summary": "Total Sales Amount: 10000, Total Units Sold: 250" - } - ] - }, - """ - - content_str: str = "" - if isinstance(message["content"], str): - content_str = message["content"] - elif isinstance(message["content"], list): - content_list: Final = message["content"] - for content in content_list: - if content["type"] == "text": - content_str += content["text"] - if len(content_str) > 0: - try: - content = json.loads(content_str) - except json.JSONDecodeError: - content = {"result": content_str} - else: - content = {} - name = "" - arguments = {} - # Recover name from last message with tool calls - if len(tool_calls) > 0: - tools: Final = tool_calls - msg_tool_call_id: Final = message.get("tool_call_id", None) - for tool in tools: - prev_tool_call_id = tool.get("id", None) - if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: - name = tool.get("function", {}).get("name", "") - arguments_str = tool.get("function", {}).get("arguments", "") - if arguments_str is not None and len(arguments_str) > 0: - arguments = json.loads(arguments_str) - - if message["role"] == "function": - function_message: Final[ChatCompletionFunctionMessage] = message - name = function_message["name"] - cohere_tool_result: ToolResultObject = { - "call": CallObject(name=name, parameters=arguments), - "outputs": [content], - } - return cohere_tool_result - else: - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - - cohere_tool_result = { - "call": CallObject(name=name, parameters=arguments), - "outputs": [content], - } - return cohere_tool_result - - -def get_all_tool_calls(messages: list) -> list: - """ - Returns extracted list of `tool_calls`. - - Done to handle openai no longer returning tool call 'name' in tool results. - """ - tool_calls: Final[list] = [] - for m in messages: - if m.get("tool_calls", None) is not None: - if isinstance(m["tool_calls"], list): - tool_calls.extend(m["tool_calls"]) - - return tool_calls - - -def convert_to_cohere_tool_invoke(tool_calls: list) -> list[ToolCallObject]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - - """ - Cohere tool invokes: - { - "role": "CHATBOT", - "tool_calls": [{"name": "get_weather", "parameters": {"location": "San Francisco, CA"}}] - } - """ - - cohere_tool_invoke: Final[list[ToolCallObject]] = [ - { - "name": get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), - "parameters": json.loads(get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments")), - } - for tool in tool_calls - if get_attribute_or_key(tool, "type") == "function" - ] - - return cohere_tool_invoke - - -def cohere_messages_pt_v2( - messages: list, - model: str, - llm_provider: str, -) -> tuple[str | ToolResultObject, ChatHistory]: - """ - Returns a tuple(Union[tool_result, message], chat_history) - - - if last message is tool result -> return 'tool_result' - - if last message is text -> return message (str) - - - return preceding messages as 'chat_history' - - Note: - - cannot specify message if the last entry in chat history contains tool results - - message must be at least 1 token long or tool results must be specified. - - cannot specify tool_results if the last entry in chat history contains a user message - """ - tool_calls: Final[list] = get_all_tool_calls(messages=messages) - - ## GET MOST RECENT MESSAGE - most_recent_message: Final = messages.pop(-1) - returned_message: ToolResultObject | str = "" - if most_recent_message.get("role", "") is not None and most_recent_message["role"] == "tool": - # tool result - returned_message = convert_openai_message_to_cohere_tool_result(most_recent_message, tool_calls) - else: - content: Final[str | list] = most_recent_message.get("content") - if isinstance(content, str): - returned_message = content - else: - for chunk in content: - if chunk.get("type") == "text": - returned_message += chunk.get("text") - - ## CREATE CHAT HISTORY - user_message_types: Final = {"user"} - tool_message_types: Final = {"tool", "function"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final[ChatHistory] = [] - msg_i = 0 - - while msg_i < len(messages): - user_content: str = "" - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - user_content += m["text"] - else: - user_content += messages[msg_i]["content"] - msg_i += 1 - - if len(user_content) > 0: - new_messages.append(ChatHistoryUser(role="USER", message=user_content)) - - system_content: str = "" - ## MERGE CONSECUTIVE SYSTEM CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - system_content += m["text"] - else: - system_content += messages[msg_i]["content"] - msg_i += 1 - - if len(system_content) > 0: - new_messages.append(ChatHistorySystem(role="SYSTEM", message=system_content)) - - assistant_content: str = "" - assistant_tool_calls: list[ToolCallObject] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - if messages[msg_i].get("content", None) is not None and isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - assistant_content += m["text"] - elif messages[msg_i].get("content") is not None and isinstance(messages[msg_i]["content"], str): - assistant_content += messages[msg_i]["content"] - if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion - assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["tool_calls"])) - - if messages[msg_i].get("function_call"): - assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["function_call"])) - - msg_i += 1 - - if len(assistant_content) > 0: - new_messages.append( - ChatHistoryChatBot( - role="CHATBOT", - message=assistant_content, - tool_calls=assistant_tool_calls, - ) - ) - - ## MERGE CONSECUTIVE TOOL RESULTS - tool_results: list[ToolResultObject] = [] - while msg_i < len(messages) and messages[msg_i]["role"] in tool_message_types: - tool_results.append(convert_openai_message_to_cohere_tool_result(messages[msg_i], tool_calls)) - - msg_i += 1 - - if len(tool_results) > 0: - new_messages.append(ChatHistoryToolResult(role="TOOL", tool_results=tool_results)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return returned_message, new_messages - - -def cohere_message_pt(messages: list): - tool_calls: Final[list] = get_all_tool_calls(messages=messages) - prompt = "" - tool_results: Final = [] - for message in messages: - # check if this is a tool_call result - if message["role"] == "tool": - tool_result = convert_openai_message_to_cohere_tool_result(message, tool_calls=tool_calls) - tool_results.append(tool_result) - elif message.get("content"): - prompt += message["content"] + "\n\n" - prompt = prompt.rstrip() - return prompt, tool_results - - -def amazon_titan_pt( - messages: list, -): # format - https://github.com/BerriAI/litellm/issues/1896 - """ - Amazon Titan uses 'User:' and 'Bot: in it's prompt template - """ - - class AmazonTitanConstants(Enum): - HUMAN_PROMPT = "\n\nUser: " # Assuming this is similar to Anthropic prompt formatting, since amazon titan's prompt formatting is currently undocumented - AI_PROMPT = "\n\nBot: " - - prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "user": - prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" - else: - prompt += f"{AmazonTitanConstants.AI_PROMPT.value}{message['content']}" - if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` - prompt = f"{AmazonTitanConstants.HUMAN_PROMPT.value}" + prompt - if messages[-1]["role"] != "assistant": - prompt += f"{AmazonTitanConstants.AI_PROMPT.value}" - return prompt - - -def _load_image_from_url(image_url): - try: - from PIL import Image - except Exception: - raise Exception("image conversion failed please run `pip install Pillow`") - from io import BytesIO - - try: - # Send a GET request to the image URL - client: Final = HTTPHandler(concurrent_limit=1) - response: Final[httpx.Response] = safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - # Check the response's content type to ensure it is an image - content_type: Final = response.headers.get("content-type") - if not content_type or "image" not in content_type: - raise ValueError(f"URL does not point to a valid image (content-type: {content_type})") - - # Load the image from the response content - return Image.open(BytesIO(response.content)) - - except Exception as e: - raise e - - -def _gemini_vision_convert_messages(messages: list): - """ - Converts given messages for GPT-4 Vision to Gemini format. - - Args: - messages (list): The messages to convert. Each message can be a dictionary with a "content" key. The content can be a string or a list of elements. If it is a string, it will be concatenated to the prompt. If it is a list, each element will be processed based on its type: - - If the element is a dictionary with a "type" key equal to "text", its "text" value will be concatenated to the prompt. - - If the element is a dictionary with a "type" key equal to "image_url", its "image_url" value will be added to the list of images. - - Returns: - tuple: A tuple containing the prompt (a string) and the processed images (a list of objects representing the images). - """ - - try: - # given messages for gpt-4 vision, convert them for gemini - # https://github.com/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_python.ipynb - prompt = "" - images: Final = [] - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - elif element["type"] == "image_url": - image_url = element["image_url"]["url"] - images.append(image_url) - # processing images passed to gemini - processed_images: Final = [] - for img in images: - if "https:/" in img: - # Case 1: Image from URL - image = _load_image_from_url(img) - processed_images.append(image) - - else: - try: - from PIL import Image - except Exception: - raise Exception("gemini image conversion failed please run `pip install Pillow`") - - if "base64" in img: - # Case 2: Base64 image data - import base64 - import io - - # Extract the base64 image data - base64_data = img.split("base64,")[1] - - # Decode the base64 image data - image_data = base64.b64decode(base64_data) - - # Load the image from the decoded data - image = Image.open(io.BytesIO(image_data)) - else: - # Case 3: Image filepath (e.g. temp.jpeg) given - image = Image.open(img) - processed_images.append(image) - content: Final = [prompt] + processed_images - return content - except Exception as e: - raise e - - -def gemini_text_image_pt(messages: list): - """ - { - "contents":[ - { - "parts":[ - {"text": "What is this picture?"}, - { - "inline_data": { - "mime_type":"image/jpeg", - "data": "'$(base64 -w0 image.jpg)'" - } - } - ] - } - ] - } - """ - try: - pass - except Exception: - raise Exception("Importing google.generativeai failed, please run 'pip install -q google-generativeai") - - prompt = "" - images: Final = [] - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - elif element["type"] == "image_url": - image_url = element["image_url"]["url"] - images.append(image_url) - - content: Final = [prompt] + images - return content - - -def azure_text_pt(messages: list): - prompt = "" - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - return prompt - - -###### AZURE AI ####### -def stringify_json_tool_call_content(messages: list) -> list: - """ - - - Check 'content' in tool role -> convert to dict (if not) -> stringify - - Done for azure_ai/cohere calls to handle results of a tool call - """ - - for m in messages: - if m["role"] == "tool" and isinstance(m["content"], str): - # check if content is a valid json object - try: - json.loads(m["content"]) - except json.JSONDecodeError: - m["content"] = json.dumps({"result": m["content"]}) - - return messages - - -###### AMAZON BEDROCK ####### - -from email.message import Message - -import httpx - -from litellm.types.llms.bedrock import ( - BedrockConverseReasoningContentBlock, - BedrockConverseReasoningTextBlock, - BedrockToolSpec, - SearchResultBlock, -) -from litellm.types.llms.bedrock import ContentBlock as BedrockContentBlock -from litellm.types.llms.bedrock import DocumentBlock as BedrockDocumentBlock -from litellm.types.llms.bedrock import ImageBlock as BedrockImageBlock -from litellm.types.llms.bedrock import SourceBlock as BedrockSourceBlock -from litellm.types.llms.bedrock import ToolBlock as BedrockToolBlock -from litellm.types.llms.bedrock import ToolResultBlock as BedrockToolResultBlock -from litellm.types.llms.bedrock import ( - ToolResultContentBlock as BedrockToolResultContentBlock, -) -from litellm.types.llms.bedrock import ToolUseBlock as BedrockToolUseBlock -from litellm.types.llms.bedrock import VideoBlock as BedrockVideoBlock - - -def _parse_content_type(content_type: str) -> str: - m: Final = Message() - m["content-type"] = content_type - return m.get_content_type() - - -def _parse_mime_type(base64_data: str) -> str | None: - mime_type_match: Final = re.match(r"data:(.*?);base64", base64_data) - if mime_type_match: - return mime_type_match.group(1) - else: - return None - - -class BedrockImageProcessor: - """Handles both sync and async image processing for Bedrock conversations.""" - - @staticmethod - def _post_call_image_processing(response: httpx.Response, image_url: str = "") -> tuple[str, str]: - # Check the response's content type to ensure it is an image - content_type: str | None = response.headers.get("content-type") - - # Use helper function to infer content type with fallback logic - content_type = infer_content_type_from_url_and_content( - url=image_url, - content=response.content, - current_content_type=content_type, - ) - - content_type = _parse_content_type(content_type) - - # Convert the image content to base64 bytes - base64_bytes: Final = base64.b64encode(response.content).decode("utf-8") - - return base64_bytes, content_type - - @staticmethod - async def get_image_details_async(image_url) -> tuple[str, str]: - try: - client: Final = get_async_httpx_client( - llm_provider=httpxSpecialProvider.PromptFactory, - params={"concurrent_limit": 1}, - ) - # Send a GET request to the image URL - response: Final[httpx.Response] = await async_safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - return BedrockImageProcessor._post_call_image_processing(response, image_url) - - except Exception as e: - raise e - - @staticmethod - def get_image_details(image_url) -> tuple[str, str]: - try: - client: Final = HTTPHandler(concurrent_limit=1) - # Send a GET request to the image URL - response: Final[httpx.Response] = safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - return BedrockImageProcessor._post_call_image_processing(response, image_url) - - except Exception as e: - raise e - - @staticmethod - def _parse_base64_image(image_url: str) -> tuple[str, str, str]: - """Parse base64 encoded image data.""" - image_metadata, img_without_base_64 = image_url.split(",") - - # Extract MIME type using regular expression - mime_type_match: Final = re.match(r"data:(.*?);base64", image_metadata) - - if mime_type_match: - mime_type = mime_type_match.group(1) - mime_type = mime_type.split(";")[0] - image_format = mime_type.split("/")[1] - else: - mime_type = "image/jpeg" - image_format = "jpeg" - - return img_without_base_64, mime_type, image_format - - @staticmethod - def _validate_format(mime_type: str, image_format: str) -> str: - """Validate image format and mime type for both images and documents.""" - - supported_image_formats: Final = litellm.AmazonConverseConfig().get_supported_image_types() - supported_doc_formats: Final = litellm.AmazonConverseConfig().get_supported_document_types() - supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() - - document_types: Final = ["application", "text"] - is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) - - supported_image_and_video_formats: Final[list[str]] = supported_video_formats + supported_image_formats - - if is_document: - return BedrockImageProcessor._get_document_format( - mime_type=mime_type, supported_doc_formats=supported_doc_formats - ) - - else: - ######################################################### - # Check if image_format is an image or video - ######################################################### - if image_format not in supported_image_and_video_formats: - raise ValueError( - f"Unsupported image format: {image_format}. Supported formats: {supported_image_and_video_formats}" - ) - return image_format - - @staticmethod - def _get_document_format(mime_type: str, supported_doc_formats: list[str]) -> str: - """ - Get the document format from the mime type - - - Primary method - uses `mimetypes.guess_all_extensions` - - Fallback method - uses `get_file_extension_from_mime_type` - - Relevant Issue: https://github.com/BerriAI/litellm/issues/12260 - - `mimetypes` is not available in docker containers, so we fallback to `get_file_extension_from_mime_type` - - Args: - mime_type: The mime type of the document - supported_doc_formats: The supported document formats for the current model - - Returns: - The document format - """ - valid_extensions: list[str] | None = None - potential_extensions: Final = mimetypes.guess_all_extensions(mime_type, strict=False) - valid_extensions = [ext[1:] for ext in potential_extensions if ext[1:] in supported_doc_formats] - - # Fallback to types/files.py if mimetypes doesn't return valid extensions - ################# - # litellm runs on docker containers and `mimetypes` depends on the installed mimetypes of the OS - # we fallback to well known mime types in types/files.py if mimetypes doesn't return valid extensions - if not valid_extensions: - try: - fallback_extension: Final = get_file_extension_from_mime_type(mime_type) - if fallback_extension in supported_doc_formats: - valid_extensions = [fallback_extension] - except ValueError: - # Neither mimetypes nor files.py could handle this MIME type - # get_file_extension_from_mime_type raises ValueError if the mime type is not supported - pass - - if not valid_extensions: - raise ValueError( - f"No supported extensions for MIME type: {mime_type}. Supported formats: {supported_doc_formats}" - ) - - # Use first valid extension instead of provided image_format - return valid_extensions[0] - - @staticmethod - def _create_bedrock_block(image_bytes: str, mime_type: str, image_format: str) -> BedrockContentBlock: - """Create appropriate Bedrock content block based on mime type.""" - _blob: Final = BedrockSourceBlock(bytes=image_bytes) - - document_types: Final = ["application", "text"] - is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) - - supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() - is_video: Final = any(image_format.startswith(video_type) for video_type in supported_video_formats) - - HASH_SAMPLE_BYTES: Final = 64 * 1024 # hash up to 64 KB of data - - if is_document: - # --- Prepare normalized bytes for hashing (without modifying original) --- - if isinstance(image_bytes, str): - # Remove whitespace/newlines so base64 variations hash identically - normalized = "".join(image_bytes.split()).encode("utf-8") - else: - normalized = image_bytes - - # --- Use only the first 64 KB for speed --- - if len(normalized) <= HASH_SAMPLE_BYTES: - sample = normalized - else: - sample = normalized[:HASH_SAMPLE_BYTES] - - # --- Compute deterministic hash (sample + total length) --- - hasher: Final = hashlib.sha256() - hasher.update(sample) - hasher.update(str(len(normalized)).encode("utf-8")) # include full length for uniqueness - full_hash: Final = hasher.hexdigest() - content_hash: Final = full_hash[:16] # short deterministic ID - - document_name: Final = f"DocumentPDFmessages_{content_hash}_{image_format}" - - # --- Return content block --- - return BedrockContentBlock( - document=BedrockDocumentBlock( - source=_blob, - format=image_format, - name=document_name, - ) - ) - elif is_video: - return BedrockContentBlock(video=BedrockVideoBlock(source=_blob, format=image_format)) - else: - return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) - - @classmethod - def process_image_sync(cls, image_url: str, format: str | None = None) -> BedrockContentBlock: - """Synchronous image processing.""" - - if "base64" in image_url: - img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) - elif "http://" in image_url or "https://" in image_url: - img_bytes, mime_type = BedrockImageProcessor.get_image_details(image_url) - image_format = mime_type.split("/")[1] - else: - raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") - - if format: - mime_type = format - image_format = mime_type.split("/")[1] - - image_format = cls._validate_format(mime_type, image_format) - return cls._create_bedrock_block(img_bytes, mime_type, image_format) - - @classmethod - async def process_image_async(cls, image_url: str, format: str | None) -> BedrockContentBlock: - """Asynchronous image processing.""" - - if "base64" in image_url: - img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) - elif "http://" in image_url or "https://" in image_url: - img_bytes, mime_type = await BedrockImageProcessor.get_image_details_async(image_url) - image_format = mime_type.split("/")[1] - else: - raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") - - if format: # override with user-defined params - mime_type = format - image_format = mime_type.split("/")[1] - - image_format = cls._validate_format(mime_type, image_format) - return cls._create_bedrock_block(img_bytes, mime_type, image_format) - - -def _convert_to_bedrock_tool_call_invoke( - tool_calls: list, - model: str | None = None, -) -> list[BedrockContentBlock]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - """ - Bedrock tool invokes: - [ - { - "role": "assistant", - "toolUse": { - "input": {"location": "Boston, MA", ..}, - "name": "get_current_weather", - "toolUseId": "call_abc123" - } - } - ] - """ - """ - - json.loads argument - - extract name - - extract id - """ - from litellm.litellm_core_utils.prompt_templates.common_utils import ( - split_concatenated_json_objects, - ) - - try: - _parts_list: Final[list[BedrockContentBlock]] = [] - for tool in tool_calls: - if "function" in tool: - tool_id = tool["id"] - name = make_valid_bedrock_tool_name(tool["function"].get("name", "")) - arguments = tool["function"].get("arguments", "") - - if not arguments or not arguments.strip(): - arguments_dict = {} - else: - try: - arguments_dict = json.loads(arguments) - # Ensure arguments_dict is always a dict - # (Bedrock requires toolUse.input to be an object). - # Some providers return arguments: '""' which - # json.loads decodes to a bare string. - if not isinstance(arguments_dict, dict): - arguments_dict = {} - except json.JSONDecodeError: - # The model may return multiple JSON objects - # concatenated in a single arguments string, e.g. - # '{"cmd":"a"}{"cmd":"b"}{"cmd":"c"}' - # Split them and emit one toolUse block per object. - # Fixes: https://github.com/BerriAI/litellm/issues/20543 - parsed_objects = split_concatenated_json_objects(arguments) - if parsed_objects: - # First object keeps the original tool id. - for obj_idx, obj in enumerate(parsed_objects): - block_id = _sanitize_bedrock_tool_use_id( - tool_id if obj_idx == 0 else f"{tool_id}_{obj_idx}" - ) - bedrock_tool = BedrockToolUseBlock(input=obj, name=name, toolUseId=block_id) - _parts_list.append(BedrockContentBlock(toolUse=bedrock_tool)) - # cache_control applies to the whole original - # tool call; attach after the last split block. - if tool.get("cache_control", None) is not None: - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool["cache_control"]}, - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts_list.append(_cache_point_block) - continue - # Fallback: no objects extracted — use empty dict. - arguments_dict = {} - - bedrock_tool = BedrockToolUseBlock( - input=arguments_dict, name=name, toolUseId=_sanitize_bedrock_tool_use_id(tool_id) - ) - bedrock_content_block = BedrockContentBlock(toolUse=bedrock_tool) - _parts_list.append(bedrock_content_block) - - # Check for cache_control and add a separate cachePoint block - if tool.get("cache_control", None) is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool["cache_control"]}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - _parts_list.append(cache_point_block) - return _parts_list - except Exception as e: - tool_call_ids: Final = tuple(tool.get("id") for tool in tool_calls if isinstance(tool, dict)) - raise litellm.BadRequestError( - message=f"Unable to convert openai tool calls with ids={tool_call_ids} to bedrock tool calls. " - f"Received error={e}", - model=model or "", - llm_provider="bedrock", - ) from e - - -def _append_bedrock_tool_result_media_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - processed_block: BedrockContentBlock, - content: dict, - content_type: str, -) -> None: - if "image" in processed_block: - tool_result_content_blocks.append(BedrockToolResultContentBlock(image=processed_block["image"])) - elif "document" in processed_block: - tool_result_content_blocks.append(BedrockToolResultContentBlock(document=processed_block["document"])) - else: - verbose_logger.warning( - "Bedrock Converse: unrecognized BedrockContentBlock keys %s for %s tool-result block %s; dropping.", - list(processed_block.keys()), - content_type, - content, - ) - - -def _append_bedrock_tool_result_image_url_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - content: dict, -) -> None: - format: str | None = None - if isinstance(content["image_url"], dict): - image_url = content["image_url"]["url"] - format = content["image_url"].get("format") - else: - image_url = content["image_url"] - processed_block: Final = BedrockImageProcessor.process_image_sync( - image_url=image_url, - format=format, - ) - _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "image_url") - - -def _append_bedrock_tool_result_file_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - content: dict, -) -> None: - # Match the user-message path (_process_file_message): accept either - # file_data (base64 data URI) or file_id (server-side reference / URL). - file_obj: Final = content.get("file") or {} - file_data: Final = file_obj.get("file_data") - file_id: Final = file_obj.get("file_id") - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={content}", - model="", - llm_provider="bedrock", - ) - processed_block: Final = BedrockImageProcessor.process_image_sync( - image_url=cast(str, file_id or file_data), - format=file_obj.get("format"), - ) - _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "file") - - -def _parse_bedrock_tool_result_content_list( - content_list: list, -) -> list[BedrockToolResultContentBlock]: - tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] - for content in content_list: - if content["type"] == "text": - tool_result_content_blocks.append(BedrockToolResultContentBlock(text=content["text"])) - elif content["type"] == "image_url": - _append_bedrock_tool_result_image_url_block(tool_result_content_blocks, content) - elif content["type"] == "file": - _append_bedrock_tool_result_file_block(tool_result_content_blocks, content) - return tool_result_content_blocks - - -def _build_bedrock_tool_result_content_blocks( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, -) -> tuple[list[BedrockToolResultContentBlock], bool]: - # Optional OpenAI tool-message extension: - # allow structured Bedrock search results on tool messages and map them - # directly to toolResult.content[].searchResult for Converse API. - # - # If `search_results` is present, we intentionally prefer it over `content` - # to avoid generating mixed text + searchResult blocks. - search_results: Final = message.get("search_results") - if isinstance(search_results, list): - tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] - for result in search_results: - if not isinstance(result, dict): - continue - tool_result_content_blocks.append( - BedrockToolResultContentBlock(searchResult=cast(SearchResultBlock, result)) - ) - if tool_result_content_blocks: - return tool_result_content_blocks, True - - message_content: Final = message["content"] - if isinstance(message_content, str): - return [BedrockToolResultContentBlock(text=message_content)], False - if isinstance(message_content, list): - return _parse_bedrock_tool_result_content_list(message_content), False - return [], False - - -def _convert_to_bedrock_tool_call_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, -) -> BedrockContentBlock: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - """ - Bedrock result looks like this: - { - "role": "user", - "content": [ - { - "toolResult": { - "toolUseId": "tooluse_kZJMlvQmRJ6eAyJE5GIl7Q", - "content": [ - { - "json": { - "song": "Elemental Hotel", - "artist": "8 Storey Hike" - } - } - ] - } - } - ] - } - """ - """ - - - """ - tool_result_content_blocks, used_search_results = _build_bedrock_tool_result_content_blocks(message) - - message.get("name", "") - id: Final = _sanitize_bedrock_tool_use_id(str(message.get("tool_call_id", str(uuid.uuid4())))) - - tool_result: Final = BedrockToolResultBlock(content=tool_result_content_blocks, toolUseId=id) - if used_search_results: - tool_result["status"] = cast(Literal["success"], "success") - - content_block: Final = BedrockContentBlock(toolResult=tool_result) - - return content_block - - -def _deduplicate_bedrock_content_blocks( - blocks: list[BedrockContentBlock], - block_key: str, - id_key: str = "toolUseId", -) -> list[BedrockContentBlock]: - """ - Remove duplicate content blocks that share the same ID under ``block_key``. - - Bedrock requires all toolResult and toolUse IDs within a single message to - be unique. When merging consecutive messages, duplicates can occur if the - same tool_call_id appears multiple times in conversation history. - - When duplicates exist, the first occurrence is retained and subsequent ones - are discarded. A warning is logged for every dropped block so that - upstream duplication bugs remain visible. - - Blocks that do not contain ``block_key`` (e.g., cachePoint, text) are - always preserved. - - Args: - blocks: The list of Bedrock content blocks to deduplicate. - block_key: The dict key to inspect (e.g. ``"toolResult"`` or ``"toolUse"``). - id_key: The nested key that holds the unique ID (default ``"toolUseId"``). - """ - seen_ids: Final[set[str]] = set() - deduplicated: Final[list[BedrockContentBlock]] = [] - for block in blocks: - keyed = block.get(block_key) - if keyed is not None and isinstance(keyed, dict): - block_id = keyed.get(id_key) - if block_id: - if block_id in seen_ids: - verbose_logger.warning( - "Bedrock Converse: dropping duplicate %s block with " - "%s=%s. This may indicate duplicate tool messages in " - "conversation history.", - block_key, - id_key, - block_id, - ) - continue - seen_ids.add(block_id) - deduplicated.append(block) - return deduplicated - - -def _deduplicate_bedrock_tool_content( - tool_content: list[BedrockContentBlock], -) -> list[BedrockContentBlock]: - """Convenience wrapper: deduplicate ``toolResult`` blocks by ``toolUseId``.""" - return _deduplicate_bedrock_content_blocks(tool_content, "toolResult") - - -def _rename_duplicate_bedrock_document_names( - contents: list[BedrockMessageBlock], -) -> list[BedrockMessageBlock]: - """ - Rename duplicate document names across all messages in a Bedrock request. - - Document names are derived from a content hash, so the same file appearing - in multiple conversation turns produces identical names and Bedrock rejects - the request with "Messages can not contain duplicate document names". The - first occurrence keeps its original name so prompt-cache prefixes stay - stable; later occurrences get a deterministic positional suffix - (``_2``, ``_3``, ...), bumped further if the suffixed name already - belongs to another document (e.g. an organic name ending in ``_2``). - """ - used_names: Final[set[str]] = set() - for message in contents: - for block in message.get("content") or []: - document = block.get("document") - if isinstance(document, dict) and document.get("name"): - used_names.add(document["name"]) - - name_counts: Final[dict[str, int]] = {} - for message in contents: - for block in message.get("content") or []: - document = block.get("document") - if not isinstance(document, dict): - continue - name = document.get("name") - if not name: - continue - count = name_counts.get(name, 0) + 1 - name_counts[name] = count - if count > 1: - suffix = count - new_name = f"{name}_{suffix}" - while new_name in used_names: - suffix += 1 - new_name = f"{name}_{suffix}" - used_names.add(new_name) - document["name"] = new_name - return contents - - -BEDROCK_DOCUMENT_PLACEHOLDER_TEXT: Final = "." - - -def _with_text_when_document_only(message: BedrockMessageBlock) -> BedrockMessageBlock: - blocks: Final = message["content"] - needs_text: Final = ( - message["role"] == "user" - and any("document" in block for block in blocks) - and all("text" not in block for block in blocks) - ) - if not needs_text: - return message - placeholder: Final = BedrockContentBlock(text=BEDROCK_DOCUMENT_PLACEHOLDER_TEXT) - cut: Final = len(blocks) - 1 if "cachePoint" in blocks[-1] else len(blocks) - return BedrockMessageBlock(role="user", content=[*blocks[:cut], placeholder, *blocks[cut:]]) - - -def _ensure_document_messages_have_text( - contents: list[BedrockMessageBlock], -) -> list[BedrockMessageBlock]: - """ - Bedrock Converse rejects any user message that carries a document block - without a sibling text block ("A text block must be included when using - documents"), e.g. Claude Code sends the PDF as a document-only user turn. - Inject a placeholder text block, kept ahead of a trailing cachePoint so - the caller's cache boundary stays the final block. - """ - return [_with_text_when_document_only(message) for message in contents] - - -def _sort_bedrock_assistant_content_blocks( - blocks: list[BedrockContentBlock], -) -> list[BedrockContentBlock]: - """ - Sort assistant content blocks so that ``text`` blocks appear before - ``toolUse`` blocks. - - Bedrock requires all ``text`` blocks to precede any ``toolUse`` blocks - within an assistant message. When the Responses API converts - function_call items before message items, the resulting ``toolUse`` - blocks can end up before ``text`` blocks, causing Bedrock to reject - the request with a 400 error because the ``toolUse`` → ``toolResult`` - pairing is broken by the intervening ``text`` block. - - Sort order (stable): - 0 - reasoningContent - 1 - text / image / document / video / other non-tool blocks - 2 - toolUse - """ - - def _sort_key(block: BedrockContentBlock) -> int: - if "reasoningContent" in block: - return 0 - if "toolUse" in block: - return 2 - if "cachePoint" in block: - # cachePoint blocks are paired with their preceding toolUse block. - # Same key as toolUse so Python's stable sort keeps them together. - return 2 - return 1 - - return sorted(blocks, key=_sort_key) - - -def _insert_assistant_continue_message( - messages: list[BedrockMessageBlock], - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> list[BedrockMessageBlock]: - """ - Add dummy message between user/tool result blocks. - - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 - """ - if assistant_continue_message is not None: - if isinstance(assistant_continue_message, str): - messages.append( - BedrockMessageBlock( - role="assistant", - content=[BedrockContentBlock(text=assistant_continue_message)], - ) - ) - elif isinstance(assistant_continue_message, dict): - text = convert_content_list_to_str(assistant_continue_message) - messages.append( - BedrockMessageBlock( - role="assistant", - content=[BedrockContentBlock(text=text)], - ) - ) - elif litellm.modify_params: - text = convert_content_list_to_str(cast(ChatCompletionAssistantMessage, DEFAULT_ASSISTANT_CONTINUE_MESSAGE)) - messages.append( - BedrockMessageBlock( - role="assistant", - content=[ - BedrockContentBlock(text=text), - ], - ) - ) - return messages - - -def get_user_message_block_or_continue_message( - message: ChatCompletionUserMessage, - user_continue_message: ChatCompletionUserMessage | None = None, -) -> ChatCompletionUserMessage: - """ - Returns the user content block - if content block is an empty string, then return the default continue message - - Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 - """ - content_block: Final = message.get("content", None) - - # Handle None case - if content_block is None or (user_continue_message is None and litellm.modify_params is False): - return skip_empty_text_blocks(message=message) - - # Handle string case - if isinstance(content_block, str): - # check if content is empty - if content_block.strip(): - return message - else: - return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) - - # Handle list case - if isinstance(content_block, list): - """ - CHECK FOR - "content": [ - { - "type": "text", - "text": "" - } - ], - """ - if not content_block: - return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) - # Create a copy of the message to avoid modifying the original - modified_content_block: Final = content_block.copy() - - for item in modified_content_block: - # Check if the list is empty - if item["type"] == "text": - if not item["text"].strip(): - # Replace empty text with continue message - _user_continue_message = ChatCompletionUserMessage( - **(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE) - ) - text = convert_content_list_to_str(_user_continue_message) - item["text"] = text - break - modified_message: Final = message.copy() - modified_message["content"] = modified_content_block - return modified_message - - # Handle unsupported type - raise ValueError(f"Unsupported content type: {type(content_block)}") - - -def return_assistant_continue_message( - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - if assistant_continue_message and isinstance(assistant_continue_message, str): - return ChatCompletionAssistantMessage( - role="assistant", - content=assistant_continue_message, - ) - elif assistant_continue_message and isinstance(assistant_continue_message, dict): - return ChatCompletionAssistantMessage(**assistant_continue_message) - else: - return DEFAULT_ASSISTANT_CONTINUE_MESSAGE - - -def _skip_empty_dict_blocks(blocks: list[dict]) -> list[dict]: - """ - Filter out empty text blocks from a list of dictionaries. - - Args: - blocks: List of dictionaries representing message content blocks - - Returns: - Filtered list of non-empty text blocks - """ - return [item for item in blocks if not (item.get("type") == "text" and not item.get("text", "").strip())] - - -@overload -def skip_empty_text_blocks( - message: ChatCompletionAssistantMessage, -) -> ChatCompletionAssistantMessage: - pass - - -@overload -def skip_empty_text_blocks( - message: ChatCompletionUserMessage, -) -> ChatCompletionUserMessage: - pass - - -def skip_empty_text_blocks( - message: ChatCompletionAssistantMessage | ChatCompletionUserMessage, -) -> ChatCompletionAssistantMessage | ChatCompletionUserMessage: - """ - Skips empty text blocks in message content text blocks. - - Do not insert content here. This is a helper function, which can also be used in base case. - """ - content_block: Final = message.get("content", None) - if content_block is None: - return message - if ( - isinstance(content_block, str) - and not content_block.strip() - and is_non_content_values_set(message) - and message["role"] == "assistant" - ): - modified_message = message.copy() - modified_message["content"] = None # user message content cannot be None - return modified_message - elif isinstance(content_block, list): - modified_content_block: Final = _skip_empty_dict_blocks(cast(list[dict], content_block)) - - # If no content remains and it's an assistant message, set content to None - if not modified_content_block and message["role"] == "assistant": - modified_message = message.copy() - modified_message["content"] = None - return modified_message - - modified_message_alt: Final = message.copy() - - # Type-specific casting based on message role - if message["role"] == "assistant": - modified_message_alt["content"] = cast( - list[OpenAIMessageContentListBlock] | None, - modified_content_block or None, - ) - elif message["role"] == "user" and modified_content_block is not None: - modified_message_alt["content"] = cast(list[ChatCompletionTextObject] | None, modified_content_block) - - return modified_message_alt - - return message - - -def process_empty_text_blocks( - message: ChatCompletionAssistantMessage, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - modified_content_block = message.get("content", None) - ## BASE CASE ## - if modified_content_block is None or not isinstance(modified_content_block, list): - return message - - # Check if all items are empty text blocks - if all(item["type"] == "text" and not item["text"].strip() for item in modified_content_block): - # Replace with a single continue message - _assistant_continue_message: Final = return_assistant_continue_message(assistant_continue_message) - modified_content_block = [ - { - "type": "text", - "text": convert_content_list_to_str(_assistant_continue_message), - } - ] - else: - # Filter out only empty text blocks, keeping non-empty text and other block types - modified_content_block = [ - item for item in modified_content_block if not (item["type"] == "text" and not item["text"].strip()) - ] - - modified_message: Final = message.copy() - modified_message["content"] = cast( - list[ChatCompletionTextObject] | list[ChatCompletionThinkingBlock], - modified_content_block, - ) - return modified_message - - -def get_assistant_message_block_or_continue_message( - message: ChatCompletionAssistantMessage, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - """ - Returns the user content block - if content block is an empty string, then return the default continue message - - Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 - """ - content_block: Final = message.get("content", None) - - # Handle Base case - if content_block is None or (assistant_continue_message is None and litellm.modify_params is False): - return skip_empty_text_blocks(message=message) - - # Handle string case - if isinstance(content_block, str): - # check if content is empty - if content_block.strip(): - return message - else: - if is_non_content_values_set(message): - modified_message: Final = message.copy() - modified_message["content"] = None - return modified_message - return return_assistant_continue_message(assistant_continue_message) - - # Handle list case - if isinstance(content_block, list): - """ - CHECK FOR - "content": [ - { - "type": "text", - "text": "" - } - ], - """ - return process_empty_text_blocks(message=message, assistant_continue_message=assistant_continue_message) - - # Handle unsupported type - raise ValueError(f"Unsupported content type: {type(content_block)}") - - -class BedrockConverseMessagesProcessor: - @staticmethod - def _initial_message_setup( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - ) -> list: - # gracefully handle base case of no messages at all - if len(messages) == 0: - if user_continue_message is not None: - messages.append(user_continue_message) - elif litellm.modify_params: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE) - else: - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + "bedrock requires at least one non-system message", - model=model, - llm_provider=llm_provider, - ) - - # if initial message is assistant message - if messages[0].get("role") is not None and messages[0]["role"] == "assistant": - if not messages[0].get("prefix"): - if user_continue_message is not None: - messages.insert(0, user_continue_message) - elif litellm.modify_params: - messages.insert(0, DEFAULT_USER_CONTINUE_MESSAGE) - - # if final message is assistant message - if messages[-1].get("role") is not None and messages[-1]["role"] == "assistant": - if not messages[-1].get("prefix"): - if user_continue_message is not None: - messages.append(user_continue_message) - elif litellm.modify_params: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE) - return messages - - @staticmethod - async def _bedrock_converse_messages_pt_async( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, - ) -> list[BedrockMessageBlock]: - contents: list[BedrockMessageBlock] = [] - msg_i = 0 - - messages = BedrockConverseMessagesProcessor._initial_message_setup( - messages, model, llm_provider, user_continue_message - ) - - while msg_i < len(messages): - user_content: list[BedrockContentBlock] = [] - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "user": - message_block = get_user_message_block_or_continue_message( - message=messages[msg_i], - user_continue_message=user_continue_message, - ) - if isinstance(message_block["content"], list): - _parts: list[BedrockContentBlock] = [] - for element in message_block["content"]: - if isinstance(element, dict): - if element["type"] == "text": - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "guarded_text": - # Wrap guarded_text in guardContent block - _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) - _parts.append(_part) - elif element["type"] in ("grounding_source", "query"): - # Contextual grounding tags are guardrail metadata; the - # model only needs the underlying text, so render them - # as plain text on the generate path. - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "image_url": - format: str | None = None - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - format = element["image_url"].get("format") - else: - image_url = element["image_url"] - _part = await BedrockImageProcessor.process_image_async( - image_url=image_url, format=format - ) - _parts.append(_part) - elif element["type"] == "file": - _part = await BedrockConverseMessagesProcessor._async_process_file_message( - message=cast(ChatCompletionFileObject, element) - ) - _parts.append(_part) - elif element["type"] == "document": - _part = BedrockConverseMessagesProcessor._process_document_message(element) - _parts.append(_part) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - if ( - _parts - and message_block.get("cache_control") is not None - and not any("cachePoint" in part for part in _parts) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - user_content.extend(_parts) - elif message_block["content"] and isinstance(message_block["content"], str): - _part = BedrockContentBlock(text=messages[msg_i]["content"]) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - user_content.append(_part) - if _cache_point_block is not None: - user_content.append(_cache_point_block) - - msg_i += 1 - if user_content: - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=user_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(user_content) - else: - contents.append(BedrockMessageBlock(role="user", content=user_content)) - - ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## - tool_content: list[BedrockContentBlock] = [] - while msg_i < len(messages) and messages[msg_i]["role"] == "tool": - current_message = messages[msg_i] - tool_call_result = _convert_to_bedrock_tool_call_result(current_message) - tool_content.append(tool_call_result) - - # Check if we need to add a separate cachePoint block - tool_msg_cache_control = None - - # Check for message-level cache_control - if current_message.get("cache_control", None) is not None: - tool_msg_cache_control = current_message["cache_control"] - # Check for content-level cache_control in list content - elif isinstance(current_message.get("content"), list): - for content_element in current_message["content"]: - if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: - tool_msg_cache_control = content_element["cache_control"] - break - - # Add a separate cachePoint block if cache_control is present - if tool_msg_cache_control is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool_msg_cache_control}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - tool_content.append(cache_point_block) - - msg_i += 1 - # Deduplicate toolResult blocks with the same toolUseId - tool_content = _deduplicate_bedrock_tool_content(tool_content) - if tool_content: - # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(tool_content) - else: - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - assistant_content: list[BedrockContentBlock] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_message_block = get_assistant_message_block_or_continue_message( - message=messages[msg_i], - assistant_continue_message=assistant_continue_message, - ) - assistant_message_content_start: int = len(assistant_content) - _assistant_content = assistant_message_block.get("content", None) - thinking_blocks = cast( - list[ChatCompletionThinkingBlock] | None, - assistant_message_block.get("thinking_blocks"), - ) - - if thinking_blocks is not None: - converted_thinking_blocks = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks - ) - ) - assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=converted_thinking_blocks, - assistant_parts=assistant_content, - ) - - if _assistant_content is not None and isinstance(_assistant_content, list): - assistants_parts: list[BedrockContentBlock] = [] - for element in _assistant_content: - if isinstance(element, dict): - if element["type"] == "thinking": - thinking_block = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] - ) - assistants_parts = ( - BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=thinking_block, - assistant_parts=assistants_parts, - ) - ) - elif element["type"] == "text": - # Skip completely empty strings to avoid blank content blocks - if element.get("text", "").strip(): - assistants_part = BedrockContentBlock(text=element["text"]) - assistants_parts.append(assistants_part) - elif element["type"] == "image_url": - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - else: - image_url = element["image_url"] - assistants_part = await BedrockImageProcessor.process_image_async(image_url=image_url) - assistants_parts.append(assistants_part) - # Add cache point block for assistant content elements - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - assistants_parts.append(_cache_point_block) - assistant_content.extend(assistants_parts) - elif _assistant_content is not None and isinstance(_assistant_content, str): - # Skip completely empty strings to avoid blank content blocks - if _assistant_content.strip(): - assistant_content.append(BedrockContentBlock(text=_assistant_content)) - # If content is empty/whitespace, skip it (don't add a placeholder) - # Add cache point block for assistant string content - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - _tool_calls = assistant_message_block.get("tool_calls", []) - if _tool_calls: - assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) - - if ( - isinstance(_assistant_content, list) - and len(assistant_content) > assistant_message_content_start - and assistant_message_block.get("cache_control") is not None - and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - msg_i += 1 - - assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") - assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) - - if assistant_content: - contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) - - @staticmethod - def translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks: list[ChatCompletionThinkingBlock], - ) -> list[BedrockContentBlock]: - reasoning_content_blocks: Final[list[BedrockContentBlock]] = [] - for thinking_block in thinking_blocks: - reasoning_text = thinking_block.get("thinking") - reasoning_signature = thinking_block.get("signature") - text_block = BedrockConverseReasoningTextBlock( - text=reasoning_text or "", - ) - if reasoning_signature is not None: - text_block["signature"] = reasoning_signature - reasoning_content_block = BedrockConverseReasoningContentBlock( - reasoningText=text_block, - ) - bedrock_content_block = BedrockContentBlock(reasoningContent=reasoning_content_block) - reasoning_content_blocks.append(bedrock_content_block) - return reasoning_content_blocks - - @staticmethod - def _process_file_message(message: ChatCompletionFileObject) -> BedrockContentBlock: - file_message: Final = message.get("file") - if file_message is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="bedrock", - ) - file_data: Final = file_message.get("file_data") - file_id: Final = file_message.get("file_id") - - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={message}", - model="", - llm_provider="bedrock", - ) - format: Final = file_message.get("format") - return BedrockImageProcessor.process_image_sync(image_url=cast(str, file_id or file_data), format=format) - - @staticmethod - async def _async_process_file_message( - message: ChatCompletionFileObject, - ) -> BedrockContentBlock: - file_message: Final = message.get("file") - if file_message is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="bedrock", - ) - file_data: Final = file_message.get("file_data") - file_id: Final = file_message.get("file_id") - format: Final = file_message.get("format") - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={message}", - model="", - llm_provider="bedrock", - ) - return await BedrockImageProcessor.process_image_async(image_url=cast(str, file_id or file_data), format=format) - - @staticmethod - def _process_document_message(element: dict) -> BedrockContentBlock: - """Convert a document content block to a Bedrock DocumentBlock. - - Handles the Anthropic-style document format: - {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "..."}} - """ - source: Final = element["source"] - source_type: Final = source.get("type") - if source_type != "base64": - raise ValueError( - f"Bedrock Converse only supports base64-encoded document sources, got '{source_type}'. " - "Please convert the document to base64 before sending to Bedrock." - ) - media_type: Final[str] = source["media_type"] - data: Final[str] = source["data"] - doc_format = BedrockImageProcessor._validate_format(mime_type=media_type, image_format=media_type.split("/")[1]) - - # Deterministic name using the same hashing pattern as _create_bedrock_block - HASH_SAMPLE_BYTES: Final = 64 * 1024 - normalized: Final = "".join(data.split()).encode("utf-8") - sample: Final = normalized[:HASH_SAMPLE_BYTES] - hasher: Final = hashlib.sha256() - hasher.update(sample) - hasher.update(str(len(normalized)).encode("utf-8")) - content_hash: Final = hasher.hexdigest()[:16] - document_name: Final = f"Document_{content_hash}_{doc_format}" - - return BedrockContentBlock( - document=BedrockDocumentBlock( - source=BedrockSourceBlock(bytes=data), - format=doc_format, - name=document_name, - ) - ) - - @staticmethod - def add_thinking_blocks_to_assistant_content( - thinking_blocks: list[BedrockContentBlock], - assistant_parts: list[BedrockContentBlock], - ) -> list[BedrockContentBlock]: - """ - If contains 'signature', it is a thinking block. - If missing 'signature', it is a text block - e.g. when using a non-anthropic model. - - Handle error raised by bedrock if thinking blocks are provided for a non-thinking model (e.g. nova with tool use) - - Relevant Issue: https://github.com/BerriAI/litellm/issues/9063 - """ - filtered_thinking_blocks: Final = [] - for block in thinking_blocks: - reasoning_content = block.get("reasoningContent", None) - reasoning_text = reasoning_content.get("reasoningText", None) if reasoning_content is not None else None - if reasoning_text and not reasoning_text.get("signature"): - reasoning_text_text = reasoning_text["text"] - if reasoning_text_text.strip(): - assistants_part = BedrockContentBlock(text=reasoning_text_text) - assistant_parts.append(assistants_part) - else: - filtered_thinking_blocks.append(block) - if len(filtered_thinking_blocks) > 0: - assistant_parts.extend(filtered_thinking_blocks) - return assistant_parts - - -def _bedrock_converse_messages_pt( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> list[BedrockMessageBlock]: - """ - Converts given messages from OpenAI format to Bedrock format - - - Roles must alternate b/w 'user' and 'model' (same as anthropic -> merge consecutive roles) - - Please ensure that function response turn comes immediately after a function call turn - - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 - """ - - contents: list[BedrockMessageBlock] = [] - msg_i = 0 - - messages = BedrockConverseMessagesProcessor._initial_message_setup( - messages, model, llm_provider, user_continue_message - ) - - while msg_i < len(messages): - user_content: list[BedrockContentBlock] = [] - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "user": - message_block = get_user_message_block_or_continue_message( - message=messages[msg_i], - user_continue_message=user_continue_message, - ) - if isinstance(message_block["content"], list): - _parts: list[BedrockContentBlock] = [] - for element in message_block["content"]: - if isinstance(element, dict): - if element["type"] == "text": - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "guarded_text": - # Wrap guarded_text in guardContent block - _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) - _parts.append(_part) - elif element["type"] in ("grounding_source", "query"): - # Contextual grounding tags are guardrail metadata; the - # model only needs the underlying text, so render them as - # plain text on the generate path. - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "image_url": - format: str | None = None - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - format = element["image_url"].get("format") - else: - image_url = element["image_url"] - _part = BedrockImageProcessor.process_image_sync( - image_url=image_url, - format=format, - ) - _parts.append(_part) - elif element["type"] == "file": - _part = BedrockConverseMessagesProcessor._process_file_message( - message=cast(ChatCompletionFileObject, element) - ) - _parts.append(_part) - elif element["type"] == "document": - _part = BedrockConverseMessagesProcessor._process_document_message(element) - _parts.append(_part) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - if ( - _parts - and message_block.get("cache_control") is not None - and not any("cachePoint" in part for part in _parts) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - user_content.extend(_parts) - elif message_block["content"] and isinstance(message_block["content"], str): - _part = BedrockContentBlock(text=messages[msg_i]["content"]) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - user_content.append(_part) - if _cache_point_block is not None: - user_content.append(_cache_point_block) - - msg_i += 1 - if user_content: - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=user_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(user_content) - else: - contents.append(BedrockMessageBlock(role="user", content=user_content)) - - ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## - tool_content: list[BedrockContentBlock] = [] - while msg_i < len(messages) and messages[msg_i]["role"] == "tool": - tool_call_result = _convert_to_bedrock_tool_call_result(messages[msg_i]) - current_message = messages[msg_i] - - # Add the tool result first - tool_content.append(tool_call_result) - - # Check if we need to add a separate cachePoint block - tool_msg_cache_control = None - - # Check for message-level cache_control - if current_message.get("cache_control", None) is not None: - tool_msg_cache_control = current_message["cache_control"] - # Check for content-level cache_control in list content - elif isinstance(current_message.get("content"), list): - for content_element in current_message["content"]: - if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: - tool_msg_cache_control = content_element["cache_control"] - break - - # Add a separate cachePoint block if cache_control is present - if tool_msg_cache_control is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool_msg_cache_control}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - tool_content.append(cache_point_block) - - msg_i += 1 - # Deduplicate toolResult blocks with the same toolUseId - tool_content = _deduplicate_bedrock_tool_content(tool_content) - if tool_content: - # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(tool_content) - else: - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - assistant_content: list[BedrockContentBlock] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_message_block = get_assistant_message_block_or_continue_message( - message=messages[msg_i], - assistant_continue_message=assistant_continue_message, - ) - assistant_message_content_start: int = len(assistant_content) - _assistant_content = assistant_message_block.get("content", None) - thinking_blocks = cast( - list[ChatCompletionThinkingBlock] | None, - assistant_message_block.get("thinking_blocks"), - ) - - if thinking_blocks is not None: - converted_thinking_blocks = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks - ) - ) - assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=converted_thinking_blocks, - assistant_parts=assistant_content, - ) - - if _assistant_content is not None and isinstance(_assistant_content, list): - assistants_parts: list[BedrockContentBlock] = [] - for element in _assistant_content: - if isinstance(element, dict): - if element["type"] == "thinking": - thinking_block = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] - ) - ) - assistants_parts = ( - BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=thinking_block, - assistant_parts=assistants_parts, - ) - ) - elif element["type"] == "text": - # AWS Bedrock doesn't allow empty or whitespace-only text content - # Skip completely empty strings to avoid blank content blocks - if element.get("text", "").strip(): - assistants_part = BedrockContentBlock(text=element["text"]) - assistants_parts.append(assistants_part) - elif element["type"] == "image_url": - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - else: - image_url = element["image_url"] - assistants_part = BedrockImageProcessor.process_image_sync(image_url=image_url) - assistants_parts.append(assistants_part) - # Add cache point block for assistant content elements - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - assistants_parts.append(_cache_point_block) - assistant_content.extend(assistants_parts) - elif _assistant_content is not None and isinstance(_assistant_content, str): - # Skip completely empty strings to avoid blank content blocks - if _assistant_content.strip(): - assistant_content.append(BedrockContentBlock(text=_assistant_content)) - # Add cache point block for assistant string content - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - _tool_calls = assistant_message_block.get("tool_calls", []) - if _tool_calls: - assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) - - if ( - isinstance(_assistant_content, list) - and len(assistant_content) > assistant_message_content_start - and assistant_message_block.get("cache_control") is not None - and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - msg_i += 1 - - assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") - assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) - - if assistant_content: - contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) - - -def make_valid_bedrock_tool_name(input_tool_name: str) -> str: - """Normalize tool names to Bedrock pattern [a-zA-Z][a-zA-Z0-9_-]*.""" - - def replace_invalid(char): - if char.isalnum() or char in ("_", "-"): - return char - return "_" - - # If the string is empty, return a default valid identifier - if input_tool_name is None or len(input_tool_name) == 0: - return input_tool_name - bedrock_tool_name = copy.copy(input_tool_name) - # If it doesn't start with a letter, prepend 'a' - if not bedrock_tool_name[0].isalpha(): - bedrock_tool_name = "a" + bedrock_tool_name - - # Replace any invalid characters with underscores - valid_string: Final = "".join(replace_invalid(char) for char in bedrock_tool_name) - - if input_tool_name != valid_string: - # passed tool name was formatted to become valid - # store it internally so we can use for the response - litellm.bedrock_tool_name_mappings.set_cache(key=valid_string, value=input_tool_name) - - return valid_string - - -def add_cache_point_tool_block(tool: dict, model: str | None = None) -> BedrockToolBlock | None: - from litellm.llms.bedrock.common_utils import ( - bedrock_model_accepts_cache_points, - is_claude_4_5_on_bedrock, - ) - - cache_control: Final = tool.get("cache_control", None) - if cache_control is not None and bedrock_model_accepts_cache_points(model): - cache_point: Final = cache_control.get("type", "ephemeral") - if cache_point == "ephemeral": - cache_point_block: Final[CachePointBlock] = {"type": "default"} - if isinstance(cache_control, dict) and "ttl" in cache_control: - ttl: Final = cache_control["ttl"] - if ttl in ["5m", "1h"] and model is not None and is_claude_4_5_on_bedrock(model): - cache_point_block["ttl"] = ttl - return {"cachePoint": cache_point_block} - return None - - -def _is_bedrock_tool_block(tool: dict) -> bool: - """ - Check if a tool is already a BedrockToolBlock. - - BedrockToolBlock has one of: systemTool, toolSpec, or cachePoint. - This is used to detect tools that are already in Bedrock format - (e.g., systemTool for Nova grounding) vs OpenAI-style function tools - that need transformation. - - Args: - tool: The tool dict to check - - Returns: - True if the tool is already a BedrockToolBlock, False otherwise - - Examples: - >>> _is_bedrock_tool_block({"systemTool": {"name": "nova_grounding"}}) - True - >>> _is_bedrock_tool_block({"type": "function", "function": {...}}) - False - """ - return isinstance(tool, dict) and ("systemTool" in tool or "toolSpec" in tool or "cachePoint" in tool) - - -def _bedrock_tools_pt(tools: list, model: str | None = None) -> list[BedrockToolBlock]: - """ - OpenAI tools looks like: - tools = [ - { - "type": "function", - "function": { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, - }, - "required": ["location"], - }, - } - } - ] - """ - """ - Bedrock toolConfig looks like: - "tools": [ - { - "toolSpec": { - "name": "top_song", - "description": "Get the most popular song played on a radio station.", - "inputSchema": { - "json": { - "type": "object", - "properties": { - "sign": { - "type": "string", - "description": "The call sign for the radio station for which you want the most popular song. Example calls signs are WZPZ, and WKRP." - } - }, - "required": [ - "sign" - ] - } - } - } - } - ] - """ - from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs - from litellm.llms.bedrock.common_utils import ( - bedrock_converse_supports_strict_tools, - normalize_json_schema_custom_types_to_object, - ) - - _valid_json_schema_root_types = frozenset(("array", "boolean", "integer", "null", "number", "object", "string")) - # Only Claude on Bedrock honours strict tool schemas; other families - # (Nova, Llama, GPT-OSS) reject the strict field outright. Opus 4.7/4.8 - # also reject `strict` on Bedrock Converse (see #31582) — their validator - # maps toolSpec to the native Anthropic tool shape, which has no strict - # field, even though Anthropic's native API accepts it as a top-level key. - supports_strict_tools: Final = bool(model and bedrock_converse_supports_strict_tools(model)) - tool_block_list: Final[list[BedrockToolBlock]] = [] - for tool_idx, tool in enumerate(tools): - # Check if tool is already a BedrockToolBlock (e.g., systemTool for Nova grounding) - if _is_bedrock_tool_block(tool): - # Already a BedrockToolBlock, pass it through - tool_block_list.append(tool) - continue - - # Responses built-in tools (web_search, image_generation, namespace, tool_search, - # custom) carry neither an OpenAI "function" nor an Anthropic "input_schema" and have - # no Bedrock toolSpec equivalent; drop them instead of emitting an empty junk toolSpec. - if isinstance(tool, dict) and "function" not in tool and "input_schema" not in tool: - continue - - # OpenAI function tools, or Anthropic Messages / Claude Code ({name, input_schema, type, ...}) - if isinstance(tool, dict) and "input_schema" in tool and "function" not in tool: - parameters = copy.deepcopy(tool.get("input_schema") or {"type": "object", "properties": {}}) - raw_name = tool.get("name", "") or "" - _tool_description = tool.get("description", None) - else: - parameters = copy.deepcopy(tool.get("function", {}).get("parameters", {"type": "object", "properties": {}})) - raw_name = tool.get("function", {}).get("name", "") or "" - _tool_description = tool.get("function", {}).get("description", None) - - if not (raw_name and str(raw_name).strip()): - raw_name = f"litellm_unnamed_tool_{tool_idx}" - - # related issue: https://github.com/BerriAI/litellm/issues/5007 - # Bedrock tool names must satisfy pattern: [a-zA-Z][a-zA-Z0-9_-]* - name = make_valid_bedrock_tool_name(input_tool_name=raw_name) - if _tool_description: # bedrock doesn't accept empty "" or None descriptions - description = _tool_description - else: - description = name - - defs = parameters.pop("$defs", {}) - defs_copy = copy.deepcopy(defs) - # Expand $ref references in parameters using the definitions - # Note: We don't pre-flatten defs as that causes exponential memory growth - # with circular references (see issue #19098). unpack_defs handles nested - # refs recursively and correctly detects/skips circular references. - unpack_defs(parameters, defs_copy) - normalize_json_schema_custom_types_to_object(parameters) - if parameters.get("type") not in _valid_json_schema_root_types: - parameters["type"] = "object" - tool_block = cast( - BedrockToolBlock, - BedrockToolSpec( - name=name, - description=description, - parameters=parameters, - strict=tool.get("function", {}).get("strict", None), - supports_strict_tools=supports_strict_tools, - ), - ) - tool_block_list.append(tool_block) - - ## ADD CACHE POINT TOOL BLOCK ## - cache_point_tool_block = add_cache_point_tool_block(tool, model=model) - if cache_point_tool_block is not None: - tool_block_list.append(cache_point_tool_block) - - return tool_block_list - - -# Function call template -def function_call_prompt(messages: list, functions: list): - function_prompt = """Produce JSON OUTPUT ONLY! Adhere to this format {"name": "function_name", "arguments":{"argument_name": "argument_value"}} The following functions are available to you:""" - for function in functions: - function_prompt += f"""\n{function}\n""" - - function_added_to_prompt = False - for message in messages: - if "system" in message["role"]: - if isinstance(message["content"], str): - message["content"] += f""" {function_prompt}""" - else: - message["content"].append({"type": "text", "text": f""" {function_prompt}"""}) - function_added_to_prompt = True - - if function_added_to_prompt is False: - messages.append({"role": "system", "content": f"""{function_prompt}"""}) - - return messages - - -def response_schema_prompt(model: str, response_schema: dict) -> str: - """ - Decides if a user-defined custom prompt or default needs to be used - - Returns the prompt str that's passed to the model as a user message - """ - custom_prompt_details: dict | None = None - response_schema_as_message: Final = [{"role": "user", "content": f"{response_schema}"}] - if f"{model}/response_schema_prompt" in litellm.custom_prompt_dict: - custom_prompt_details = litellm.custom_prompt_dict[ - f"{model}/response_schema_prompt" - ] # allow user to define custom response schema prompt by model - elif "response_schema_prompt" in litellm.custom_prompt_dict: - custom_prompt_details = litellm.custom_prompt_dict["response_schema_prompt"] - - if custom_prompt_details is not None: - return custom_prompt( - role_dict=custom_prompt_details["roles"], - initial_prompt_value=custom_prompt_details["initial_prompt_value"], - final_prompt_value=custom_prompt_details["final_prompt_value"], - messages=response_schema_as_message, - ) - else: - return default_response_schema_prompt(response_schema=response_schema) - - -def default_response_schema_prompt(response_schema: dict) -> str: - """ - Used if provider/model doesn't support 'response_schema' param. - - This is the default prompt. Allow user to override this with a custom_prompt. - """ - prompt_str: Final = f"""Use this JSON schema: - ```json - {response_schema} - ```""" - return prompt_str - - -# Custom prompt template -def custom_prompt( - role_dict: dict, - messages: list, - initial_prompt_value: str = "", - final_prompt_value: str = "", - bos_token: str = "", - eos_token: str = "", -) -> str: - prompt = bos_token + initial_prompt_value - bos_open = True - ## a bos token is at the start of a system / human message - ## an eos token is at the end of the assistant response to the message - for message in messages: - role = message["role"] - - if role in ["system", "human"] and not bos_open: - prompt += bos_token - bos_open = True - - pre_message_str = ( - role_dict[role]["pre_message"] if role in role_dict and "pre_message" in role_dict[role] else "" - ) - post_message_str = ( - role_dict[role]["post_message"] if role in role_dict and "post_message" in role_dict[role] else "" - ) - if isinstance(message["content"], str): - prompt += pre_message_str + message["content"] + post_message_str - elif isinstance(message["content"], list): - text_str = "" - for content in message["content"]: - if content.get("text", None) is not None and isinstance(content["text"], str): - text_str += content["text"] - prompt += pre_message_str + text_str + post_message_str - - if role == "assistant": - prompt += eos_token - bos_open = False - - prompt += final_prompt_value - return prompt - - -def prompt_factory( - model: str, - messages: list, - custom_llm_provider: str | None = None, - api_key: str | None = None, -): - original_model_name: Final = model - model = model.lower() - if custom_llm_provider == "ollama": - return ollama_pt(model=model, messages=messages) - elif custom_llm_provider == "anthropic": - if litellm.AnthropicTextConfig._is_anthropic_text_model(model): - return anthropic_pt(messages=messages) - return anthropic_messages_pt(messages=messages, model=model, llm_provider=custom_llm_provider) - elif custom_llm_provider == "anthropic_xml": - return anthropic_messages_pt_xml(messages=messages) - elif custom_llm_provider == "gemini": - if ( - model == "gemini-pro-vision" - or litellm.supports_vision(model=model) - or litellm.supports_vision(model=custom_llm_provider + "/" + model) - ): - return _gemini_vision_convert_messages(messages=messages) - else: - return gemini_text_image_pt(messages=messages) - elif custom_llm_provider == "mistral": - return litellm.MistralConfig()._transform_messages(messages=messages, model=model) - elif custom_llm_provider == "bedrock": - if "amazon.titan-text" in model: - return amazon_titan_pt(messages=messages) - elif "anthropic." in model: - if any(_ in model for _ in ["claude-2.1", "claude-v2:1"]): - return claude_2_1_pt(messages=messages) - else: - return anthropic_pt(messages=messages) - elif "mistral." in model: - return mistral_instruct_pt(messages=messages) - elif "llama2" in model and "chat" in model: - return llama_2_chat_pt(messages=messages) - elif ("llama3" in model or "llama4" in model) and "instruct" in model: - return hf_chat_template( - model="meta-llama/Meta-Llama-3-8B-Instruct", - messages=messages, - ) - - elif custom_llm_provider == "clarifai": - if "claude" in model: - return anthropic_pt(messages=messages) - - elif custom_llm_provider == "perplexity": - for message in messages: - message.pop("name", None) - return messages - elif custom_llm_provider == "azure_text": - return azure_text_pt(messages=messages) - elif custom_llm_provider == "watsonx": - from litellm.llms.watsonx.chat.transformation import IBMWatsonXChatConfig - - return IBMWatsonXChatConfig.apply_prompt_template(model=model, messages=messages) - - try: - if "meta-llama/llama-2" in model and "chat" in model: - return llama_2_chat_pt(messages=messages) - elif ("meta-llama/llama-3" in model or "meta-llama-3" in model) and "instruct" in model: - return hf_chat_template( - model="meta-llama/Meta-Llama-3-8B-Instruct", - messages=messages, - ) - elif ( - "tiiuae/falcon" in model - ): # Note: for the instruct models, it's best to use a User: .., Assistant:.. approach in your prompt template. - if model == "tiiuae/falcon-180B-chat": - return falcon_chat_pt(messages=messages) - elif "instruct" in model: - return falcon_instruct_pt(messages=messages) - elif "mosaicml/mpt" in model: - if "chat" in model: - return mpt_chat_pt(messages=messages) - elif "codellama/codellama" in model or "togethercomputer/codellama" in model: - if "instruct" in model: - return llama_2_chat_pt( - messages=messages - ) # https://huggingface.co/blog/codellama#conversational-instructions - elif "wizardlm/wizardcoder" in model: - return wizardcoder_pt(messages=messages) - elif "phind/phind-codellama" in model: - return phind_codellama_pt(messages=messages) - elif "togethercomputer/llama-2" in model and ("instruct" in model or "chat" in model): - return llama_2_chat_pt(messages=messages) - elif model in [ - "gryphe/mythomax-l2-13b", - "gryphe/mythomix-l2-13b", - "gryphe/mythologic-l2-13b", - ]: - return alpaca_pt(messages=messages) - elif model.strip() == "meta-llama/llamaguard-7b": - chat_template = "{% if messages|length % 2 == 0 %}{% set role = 'Agent' %}{% else %}{% set role = 'User' %}{% endif %}{{ \"[INST] Task: Check if there is unsafe content in '\" + role + \"' messages in conversations according our safety policy with the below categories.\n\n\nO1: Violence and Hate. \nShould not\n- Help people plan or engage in violence.\n- Encourage people to engage in violence.\n- Express hateful, derogatory, or demeaning sentiments against people based on sensitive personal characteristics like their race, color, religion, national origin, sexual orientation, gender, gender identity, or disability.\n- Encourage discrimination against people based on those characteristics and should not use slurs or create insults based on those characteristics.\nCan\n- Provide information on violence and discrimination.\n- Discuss the topics of hate and violence, and can discuss historical events involving violence.\nO2: Sexual Content.\nShould not\n- Engage in sexually explicit (i.e., erotic) conversations and should not encourage people to engage in sexual activities.\nCan\n- Discuss topics related to sexuality and sexual education.\n- Provide general information regarding sexual activities.\n- Provide guidance on topics related to sexual health.\nO3: Criminal Planning.\nShould not\n- Help people plan or engage in criminal activities like theft, kidnapping, or financial crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in criminal activities or condone criminal behavior.\nCan\n- Help people understand how crimes are committed, provided that the response is not so detailed that it could be used to provide a roadmap for future crimes.\n- Help people understand how systems work that are used to prevent crimes, provided those descriptions do not provide detailed information on how to defeat those systems.\nO4: Guns and Illegal Weapons.\nShould not\n- Assist in the planning or execution of firearm or illegal weapons crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in firearm or illegal weapons crimes or condone such behavior.\nCan\n- Discuss firearms and the arguments for and against firearm ownership.\nO5: Regulated or Controlled Substances.\nShould not\n- Encourage or assist people in the trafficking, consumption, or use of regulated or controlled substances, including but not limited to illegal drugs, tobacco, alcohol, and cannabis.\n- Assist or encourage people to create such substances when it is illegal to do so.\nCan\n- Discuss the use of regulated or controlled substances, and can provide general information on those substances (ex: what their chemical composition is and what their effects are).\n- Provide information on how those substances are manufactured, provided that the information is not so detailed as to provide step-by-step instructions for their creation.\nO6: Self-Harm.\nShould not\n- Encourage people to harm themselves.\n- Romanticize or condone self-harm or suicide.\n- Provide information regarding the methods of suicide or self-harm.\n- Provide any information or advice, other than information regarding appropriate health resources, to people who express or imply that they wish to harm themselves.\nCan\n- Provide general information regarding self-harm and suicide (ex: statistics, the mental health challenges faced by those who take these actions).\nShould\n- Provide information regarding appropriate health resources to people who imply or express the desire to harm themselves.\n\n\n\n\n\" }}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{% set role = 'User' %}{% elif message['role'] == 'assistant' %}{% set role = 'Agent' %}{% endif %}{{ role + ': ' + content.strip() + '\n\n' }}{% endfor %}{{ \"\n\nProvide your safety assessment for \" + role + \" in the above conversation:\n- First line must read 'safe' or 'unsafe'.\n- If unsafe, a second line must include a comma-separated list of violated categories. [/INST]\" }}" - return hf_chat_template(model=model, messages=messages, chat_template=chat_template) - else: - return hf_chat_template(original_model_name, messages) - except Exception: - return default_pt( - messages=messages - ) # default that covers Bloom, T-5, any non-chat tuned model (e.g. base Llama2) - - -def get_attribute_or_key(tool_or_function, attribute, default=None): - if hasattr(tool_or_function, attribute): - return getattr(tool_or_function, attribute) - if isinstance(tool_or_function, Mapping): - return tool_or_function.get(attribute, default) - return default - - -class NormalizedToolCall(TypedDict): - id: str | None - name: str | None - arguments: dict[str, object] - - -_ArgumentObjects: TypeAlias = tuple[dict[str, object], ...] -_ParsedToolCall: TypeAlias = tuple[str | None, str | None, _ArgumentObjects] - - -def _optional_call_id(value: object) -> str | None: - if isinstance(value, str) and value: - return value - return None - - -def _optional_tool_name(value: object) -> str | None: - if isinstance(value, str): - return value - return None - - -def _split_tool_call_ids(calls: Sequence[tuple[str | None, int]]) -> tuple[tuple[str | None, ...], ...]: - taken: Final = frozenset(_sanitize_anthropic_tool_use_id(call_id) for call_id, _ in calls if call_id) - - def fresh(call_id: str) -> Iterator[str]: - return filter( - lambda candidate: _sanitize_anthropic_tool_use_id(candidate) not in taken, - (f"{call_id}__concat_{n}" for n in itertools.count(1)), - ) - - suffixes: Final = MappingProxyType( - {_sanitize_anthropic_tool_use_id(call_id): fresh(call_id) for call_id, count in calls if call_id and count > 1} - ) - return tuple( - ( - call_id, - *(next(suffixes[_sanitize_anthropic_tool_use_id(call_id)]) for _ in range(count - 1)), - ) - if call_id - else (None,) * count - for call_id, count in calls - ) - - -def _parse_tool_call_arguments(raw: object, tool_name: str | None, context: str) -> _ArgumentObjects: - # Anthropic's tool_use blocks already carry a parsed dict in "input"; - # chat completions and the Responses API carry a JSON string that may be - # truncated by the model, so route those through the repair-aware parser. - if isinstance(raw, dict): - return (raw,) - if not isinstance(raw, str): - return ({},) - normalized_raw: Final = "{}" if raw == REDACTED_BY_LITELLM else raw - try: - parsed: Final = parse_tool_call_arguments(normalized_raw, tool_name=tool_name, context=context) - except ValueError as e: - salvaged: Final = salvage_concatenated_tool_arguments(normalized_raw) - if salvaged: - verbose_logger.warning( - "Recovered %d tool call(s) from concatenated JSON arguments for tool '%s' (%s)", - len(salvaged), - tool_name or "", - context, - ) - return salvaged - verbose_logger.warning("Failed to parse tool call arguments: %s", e) - return ({},) - return (parsed,) if isinstance(parsed, dict) else ({},) - - -def _choice_tool_calls(choice: object) -> tuple[object, ...]: - message: Final = get_attribute_or_key(choice, "message", None) - tool_calls: Final = get_attribute_or_key(message, "tool_calls", None) if message is not None else None - if isinstance(tool_calls, list): - return tuple(tool_calls) - return () - - -def _selected_choices(response: object, include_all_choices: bool) -> tuple[object, ...]: - choices: Final = get_attribute_or_key(response, "choices", None) - if not isinstance(choices, list) or not choices: - return () - if include_all_choices: - return tuple(choices) - return (choices[0],) - - -def _parsed_chat_tool_call(tool_call: object) -> _ParsedToolCall | None: - function: Final = get_attribute_or_key(tool_call, "function", None) - if function is None: - return None - name: Final = _optional_tool_name(get_attribute_or_key(function, "name")) - return ( - _optional_call_id(get_attribute_or_key(tool_call, "id")), - name, - _parse_tool_call_arguments( - get_attribute_or_key(function, "arguments", "{}"), - tool_name=name, - context="chat completions", - ), - ) - - -def _parsed_calls_in_choice(choice: object) -> tuple[_ParsedToolCall, ...]: - return tuple( - parsed for tool_call in _choice_tool_calls(choice) if (parsed := _parsed_chat_tool_call(tool_call)) is not None - ) - - -def _parsed_chat_tool_calls(response: object, include_all_choices: bool) -> tuple[_ParsedToolCall, ...]: - grouped: Final = tuple( - _parsed_calls_in_choice(choice) for choice in _selected_choices(response, include_all_choices) - ) - return tuple(itertools.chain.from_iterable(grouped)) - - -def _normalized_tool_calls_for_parse( - name: str | None, - call_ids: tuple[str | None, ...], - arguments: _ArgumentObjects, -) -> tuple[NormalizedToolCall, ...]: - return tuple( - NormalizedToolCall(id=call_id, name=name, arguments=argument) - for call_id, argument in zip(call_ids, arguments, strict=True) - ) - - -def _normalized_tool_calls_from_parses(parses: Sequence[_ParsedToolCall]) -> tuple[NormalizedToolCall, ...]: - id_groups: Final = _split_tool_call_ids(tuple((call_id, len(arguments)) for call_id, _, arguments in parses)) - grouped: Final = tuple( - _normalized_tool_calls_for_parse(name, call_ids, arguments) - for (_, name, arguments), call_ids in zip(parses, id_groups, strict=True) - ) - return tuple(itertools.chain.from_iterable(grouped)) - - -def _tool_calls_from_chat_completion_response( - response: object, include_all_choices: bool = False -) -> tuple[NormalizedToolCall, ...]: - return _normalized_tool_calls_from_parses(_parsed_chat_tool_calls(response, include_all_choices)) - - -def _response_function_calls(response: object) -> tuple[object, ...]: - output: Final = get_attribute_or_key(response, "output", None) - if not isinstance(output, list): - return () - return tuple(item for item in output if get_attribute_or_key(item, "type") == "function_call") - - -def _parsed_response_tool_call(item: object) -> _ParsedToolCall: - name: Final = _optional_tool_name(get_attribute_or_key(item, "name")) - raw_id: Final = get_attribute_or_key(item, "call_id") or get_attribute_or_key(item, "id") - return ( - _optional_call_id(raw_id), - name, - _parse_tool_call_arguments( - get_attribute_or_key(item, "arguments", "{}"), - tool_name=name, - context="responses API", - ), - ) - - -def _tool_calls_from_responses_api_response(response: object) -> tuple[NormalizedToolCall, ...]: - parses: Final = tuple(_parsed_response_tool_call(item) for item in _response_function_calls(response)) - return _normalized_tool_calls_from_parses(parses) - - -def _tool_calls_from_anthropic_messages_response(response: object) -> list[NormalizedToolCall]: - content: Final = get_attribute_or_key(response, "content", None) - if not isinstance(content, list): - return [] - result: Final[list[NormalizedToolCall]] = [] - for block in content: - if get_attribute_or_key(block, "type") != "tool_use": - continue - raw_input = get_attribute_or_key(block, "input", {}) - result.append( - NormalizedToolCall( - id=get_attribute_or_key(block, "id"), - name=get_attribute_or_key(block, "name"), - arguments=raw_input if isinstance(raw_input, dict) else {}, - ) - ) - return result - - -def get_tool_calls_from_response(response: object, include_all_choices: bool = False) -> list[NormalizedToolCall]: - """ - Extract tool/function calls from a response object into a normalized - ``{"id", "name", "arguments"}`` shape, regardless of which API surface - produced it: chat completions (``choices[].message.tool_calls``), - the Responses API (``output`` items of type ``function_call``), or the - Anthropic Messages API (``content`` blocks of type ``tool_use``). - - ``include_all_choices`` decides the chat-completions scope: the default - reads only ``choices[0]``, which is what consumers that act on THE reply - (e.g. guardrails rebuilding the primary assistant message) want; usage - accounting passes True because every choice of an ``n>1`` request costs - money and its tool calls really ran. The other surfaces have a single - output, so the flag has no effect on them. - - Callers that only care about a specific tool should filter the result by - ``name`` themselves -- this returns every tool call found. - """ - chat_tool_calls: Final = _tool_calls_from_chat_completion_response( - response, include_all_choices=include_all_choices - ) - if chat_tool_calls: - return list(chat_tool_calls) - for extractor in ( - _tool_calls_from_responses_api_response, - _tool_calls_from_anthropic_messages_response, - ): - tool_calls = extractor(response) - if tool_calls: - return list(tool_calls) - return [] - - -def has_tool_with_name(tools: object, tool_name: str) -> bool: - """ - Check whether a tools list (as sent to an LLM) includes a tool with the - given name, regardless of shape: OpenAI-style function tools - (``{"type": "function", "function": {"name": ...}}``) or Anthropic's - native tool shape (a top-level ``"name"``, e.g. - ``{"name": ..., "input_schema": ...}``). Anthropic's documented client - tool format doesn't require a ``"type"`` key at all -- ``"custom"`` is - only one of several possible values -- so any non-OpenAI-shaped tool is - matched on its top-level ``"name"``. - """ - if not isinstance(tools, list): - return False - for tool in tools: - if not isinstance(tool, dict): - continue - function = tool.get("function") - if tool.get("type") == "function" and isinstance(function, dict): - if function.get("name") == tool_name: - return True - elif tool.get("name") == tool_name: - return True - return False - - -def resolve_structured_messages( - messages: list[dict[str, object]] | None, - request_kwargs: dict[str, Any], -) -> list[dict[str, object]] | None: - """ - Normalize a request's messages to OpenAI-spec chat-completions shape, - regardless of which API surface produced them (chat completions, - Anthropic /v1/messages, Responses API ``input``, etc). - - Returns ``messages`` unchanged if already present. Otherwise dispatches - through the guardrail translation handlers (the same per-surface - conversion logic guardrails use) to convert e.g. Responses API ``input`` - into a message list. Returns ``None`` if no messages could be resolved. - """ - if messages: - return messages - - from litellm.litellm_core_utils.api_route_to_call_types import ( - get_call_types_for_route, - ) - from litellm.llms import load_guardrail_translation_mappings - from litellm.types.utils import CallTypes - - mappings: Final = load_guardrail_translation_mappings() - call_type: CallTypes | None = None - - # 1. Try route-based inference from proxy metadata - route: Final = request_kwargs.get("litellm_metadata", {}).get("user_api_key_request_route") - if route: - call_types_list: Final = get_call_types_for_route(route) - if call_types_list: - for ct in call_types_list: - if ct in mappings: - call_type = ct - break - - # 2. Fallback: try each mapped handler until one produces messages - handlers_to_try: Final[list[Any]] = [] - if call_type is not None and call_type in mappings: - handlers_to_try.append(mappings[call_type]()) - else: - handlers_to_try.extend(handler_cls() for handler_cls in mappings.values()) - - for handler in handlers_to_try: - structured = handler.get_structured_messages(request_kwargs) - if structured: - return [msg if isinstance(msg, dict) else msg.model_dump() for msg in structured] - return None +import base64 +import copy +import hashlib +import itertools +import json +import mimetypes +import re +import xml.etree.ElementTree as ET +from collections.abc import Iterator, Mapping, Sequence +from enum import Enum +from types import MappingProxyType +from typing import Any, Final, TypeAlias, TypedDict, cast, overload + +from jinja2.sandbox import ImmutableSandboxedEnvironment + +import litellm +import litellm.types +import litellm.types.llms +from litellm import verbose_logger +from litellm._uuid import uuid +from litellm.constants import REDACTED_BY_LITELLM +from litellm.litellm_core_utils.prompt_templates.mid_conversation_system import anthropic_system_messages +from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get +from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client +from litellm.types.files import get_file_extension_from_mime_type +from litellm.types.llms.anthropic import * +from litellm.types.llms.bedrock import CachePointBlock +from litellm.types.llms.bedrock import MessageBlock as BedrockMessageBlock +from litellm.types.llms.custom_http import httpxSpecialProvider +from litellm.types.llms.ollama import OllamaVisionModelObject +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionAssistantMessage, + ChatCompletionAssistantToolCall, + ChatCompletionFileObject, + ChatCompletionFileObjectFile, + ChatCompletionFunctionMessage, + ChatCompletionImageObject, + ChatCompletionImageUrlObject, + ChatCompletionTextObject, + ChatCompletionToolCallFunctionChunk, + ChatCompletionToolMessage, + ChatCompletionUserMessage, + OpenAIMessageContentListBlock, +) +from litellm.types.llms.vertex_ai import FunctionCall as VertexFunctionCall +from litellm.types.llms.vertex_ai import FunctionResponse as VertexFunctionResponse +from litellm.types.llms.vertex_ai import PartType as VertexPartType +from litellm.types.utils import GenericImageParsingChunk + +from .common_utils import ( + convert_content_list_to_str, + infer_content_type_from_url_and_content, + is_non_content_values_set, + is_unsignable_thinking_block, + parse_tool_call_arguments, + salvage_concatenated_tool_arguments, +) +from .image_handling import convert_url_to_base64 + + +def default_pt(messages): + return " ".join(message["content"] for message in messages) + + +def prompt_injection_detection_default_pt(): + return """Detect if a prompt is safe to run. Return 'UNSAFE' if not.""" + + +BAD_MESSAGE_ERROR_STR: Final = "Invalid Message " + +# Separator used to embed Gemini thought signatures in tool call IDs +# See: https://ai.google.dev/gemini-api/docs/thought-signatures +THOUGHT_SIGNATURE_SEPARATOR: Final = "__thought__" + +# used to interweave user messages, to ensure user/assistant alternating +DEFAULT_USER_CONTINUE_MESSAGE: Final = { + "role": "user", + "content": "Please continue.", +} # similar to autogen. Only used if `litellm.modify_params=True`. + +DEFAULT_USER_CONTINUE_MESSAGE_TYPED: Final = ChatCompletionUserMessage( + role="user", + content="Please continue.", +) + +# used to interweave assistant messages, to ensure user/assistant alternating +DEFAULT_ASSISTANT_CONTINUE_MESSAGE: Final = ChatCompletionAssistantMessage( + role="assistant", + content=[ + { + "type": "text", + "text": "Please continue.", + } + ], +) # similar to autogen. Only used if `litellm.modify_params=True`. + + +def map_system_message_pt(messages: list) -> list: + """ + Convert 'system' message to 'user' message if provider doesn't support 'system' role. + + Enabled via `completion(...,supports_system_message=False)` + + If next message is a user message or assistant message -> merge system prompt into it + + if next message is system -> append a user message instead of the system message + """ + + new_messages: Final = [] + for i, m in enumerate(messages): + if m["role"] == "system": + if i < len(messages) - 1: # Not the last message + next_m = messages[i + 1] + next_role = next_m["role"] + if next_role == "user" or next_role == "assistant": # Next message is a user or assistant message + # Merge system prompt into the next message + next_m["content"] = m["content"] + " " + next_m["content"] + elif next_role == "system": # Next message is a system message + # Append a user message instead of the system message + new_message = {"role": "user", "content": m["content"]} + new_messages.append(new_message) + else: # Last message + new_message = {"role": "user", "content": m["content"]} + new_messages.append(new_message) + else: # Not a system message + new_messages.append(m) + + return new_messages + + +# alpaca prompt template - for models like mythomax, etc. +def alpaca_pt(messages): + prompt: Final = custom_prompt( + role_dict={ + "system": { + "pre_message": "### Instruction:\n", + "post_message": "\n\n", + }, + "user": { + "pre_message": "### Instruction:\n", + "post_message": "\n\n", + }, + "assistant": {"pre_message": "### Response:\n", "post_message": "\n\n"}, + }, + bos_token="", + eos_token="", + messages=messages, + ) + return prompt + + +# Llama2 prompt template +def llama_2_chat_pt(messages): + prompt: Final = custom_prompt( + role_dict={ + "system": { + "pre_message": "[INST] <>\n", + "post_message": "\n<>\n [/INST]\n", + }, + "user": { # follow this format https://github.com/facebookresearch/llama/blob/77062717054710e352a99add63d160274ce670c6/llama/generation.py#L348 + "pre_message": "[INST] ", + "post_message": " [/INST]\n", + }, + "assistant": { + "post_message": "\n" # follows this - https://replicate.com/blog/how-to-prompt-llama + }, + }, + messages=messages, + bos_token="", + eos_token="", + ) + return prompt + + +def convert_to_ollama_image(openai_image_url: str): + try: + if openai_image_url.startswith("http"): + openai_image_url = convert_url_to_base64(url=openai_image_url) + + if openai_image_url.startswith("data:image/"): + # Extract the base64 image data + base64_data = openai_image_url.split("data:image/")[1].split(";base64,")[1] + else: + base64_data = openai_image_url + + return base64_data + except Exception as e: + if "Error: Unable to fetch image from URL" in str(e): + raise e + raise Exception( + """Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{base64_image}". """ + ) + + +def _handle_ollama_system_message(messages: list, prompt: str, msg_i: int) -> tuple[str, int]: + system_content_str = "" + ## MERGE CONSECUTIVE SYSTEM CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + msg_content = convert_content_list_to_str(messages[msg_i]) + system_content_str += msg_content + + msg_i += 1 + + return system_content_str, msg_i + + +def ollama_pt( + model: str, messages: list +) -> ( + str | OllamaVisionModelObject +): # https://github.com/ollama/ollama/blob/af4cf55884ac54b9e637cd71dadfe9b7a5685877/docs/modelfile.md#template + user_message_types: Final = {"user", "tool", "function"} + msg_i = 0 + images: Final = [] + prompt = "" + while msg_i < len(messages): + init_msg_i = msg_i + user_content_str = "" + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + msg_content = messages[msg_i].get("content") + if msg_content: + if isinstance(msg_content, list): + for m in msg_content: + if m.get("type", "") == "image_url": + if isinstance(m["image_url"], str): + images.append(m["image_url"]) + elif isinstance(m["image_url"], dict): + images.append(m["image_url"]["url"]) + elif m.get("type", "") == "text": + user_content_str += m["text"] + else: + # Tool message content will always be a string + user_content_str += msg_content + + msg_i += 1 + + if user_content_str: + prompt += f"### User:\n{user_content_str}\n\n" + + system_content_str, msg_i = _handle_ollama_system_message(messages, prompt, msg_i) + if system_content_str: + prompt += f"### System:\n{system_content_str}\n\n" + + assistant_content_str = "" + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_content_str += convert_content_list_to_str(messages[msg_i]) + + tool_calls = messages[msg_i].get("tool_calls") + ollama_tool_calls = [] + if tool_calls: + for call in tool_calls: + call_id: str = call["id"] + function_name: str = call["function"]["name"] + arguments = json.loads(call["function"]["arguments"]) + + ollama_tool_calls.append( + { + "id": call_id, + "type": "function", + "function": { + "name": function_name, + "arguments": arguments, + }, + } + ) + + if ollama_tool_calls: + assistant_content_str += f"Tool Calls: {json.dumps(ollama_tool_calls, indent=2)}" + + msg_i += 1 + + if assistant_content_str: + prompt += f"### Assistant:\n{assistant_content_str}\n\n" + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider="ollama", + ) + + response_dict: Final[OllamaVisionModelObject] = { + "prompt": prompt, + "images": images, + } + + return response_dict + + +def mistral_instruct_pt(messages): + # Following the Mistral example's https://huggingface.co/docs/transformers/main/chat_templating + prompt: Final = custom_prompt( + initial_prompt_value="", + role_dict={ + "system": { + "pre_message": "[INST] \n", + "post_message": " [/INST]\n", + }, + "user": {"pre_message": "[INST] ", "post_message": " [/INST]\n"}, + "assistant": {"pre_message": " ", "post_message": " "}, + }, + final_prompt_value="", + messages=messages, + ) + return prompt + + +# Falcon prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 +def falcon_instruct_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += message["content"] + else: + prompt += message["role"] + ":" + message["content"].replace("\r\n", "\n").replace("\n\n", "\n") + prompt += "\n\n" + + return prompt + + +def falcon_chat_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "System: " + message["content"] + elif message["role"] == "assistant": + prompt += "Falcon: " + message["content"] + elif message["role"] == "user": + prompt += "User: " + message["content"] + + return prompt + + +# MPT prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 +def mpt_chat_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "<|im_start|>system" + message["content"] + "<|im_end|>" + "\n" + elif message["role"] == "assistant": + prompt += "<|im_start|>assistant" + message["content"] + "<|im_end|>" + "\n" + elif message["role"] == "user": + prompt += "<|im_start|>user" + message["content"] + "<|im_end|>" + "\n" + return prompt + + +# WizardCoder prompt template - https://huggingface.co/WizardLM/WizardCoder-Python-34B-V1.0#prompt-format +def wizardcoder_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += message["content"] + "\n\n" + elif message["role"] == "user": # map to 'Instruction' + prompt += "### Instruction:\n" + message["content"] + "\n\n" + elif message["role"] == "assistant": # map to 'Response' + prompt += "### Response:\n" + message["content"] + "\n\n" + return prompt + + +# Phind-CodeLlama prompt template - https://huggingface.co/Phind/Phind-CodeLlama-34B-v2#how-to-prompt-the-model +def phind_codellama_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "### System Prompt\n" + message["content"] + "\n\n" + elif message["role"] == "user": + prompt += "### User Message\n" + message["content"] + "\n\n" + elif message["role"] == "assistant": + prompt += "### Assistant\n" + message["content"] + "\n\n" + return prompt + + +def _render_chat_template(env, chat_template: str, bos_token: str, eos_token: str, messages: list) -> str: + """ + Shared template rendering logic for both sync and async hf_chat_template + + Args: + env: Jinja2 environment + chat_template: Chat template string + bos_token: Beginning of sequence token + eos_token: End of sequence token + messages: Messages to render + + Returns: + Rendered template string + """ + try: + template: Final = env.from_string(chat_template) + except Exception as e: + raise e + + def _is_system_in_template(): + try: + # Try rendering the template with a system message + template.render( + messages=[{"role": "system", "content": "test"}], + eos_token="", + bos_token="", + ) + return True + # This will be raised if Jinja attempts to render the system message and it can't + except Exception: + return False + + try: + rendered_text = "" + # Render the template with the provided values + if _is_system_in_template(): + rendered_text = template.render( + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + add_generation_prompt=True, + ) + else: + # treat a system message as a user message, if system not in template + reformatted_messages: Final = [] + try: + for message in messages: + if message["role"] == "system": + reformatted_messages.append({"role": "user", "content": message["content"]}) + else: + reformatted_messages.append(message) + rendered_text = template.render( + bos_token=bos_token, + eos_token=eos_token, + messages=reformatted_messages, + add_generation_prompt=True, + ) + except Exception as e: + if "Conversation roles must alternate user/assistant" in str(e): + # reformat messages to ensure user/assistant are alternating + new_messages: Final = [] + for i in range(len(reformatted_messages) - 1): + new_messages.append(reformatted_messages[i]) + if reformatted_messages[i]["role"] == reformatted_messages[i + 1]["role"]: + if reformatted_messages[i]["role"] == "user": + new_messages.append({"role": "assistant", "content": ""}) + else: + new_messages.append({"role": "user", "content": ""}) + new_messages.append(reformatted_messages[-1]) + rendered_text = template.render(bos_token=bos_token, eos_token=eos_token, messages=new_messages) + + return rendered_text + except Exception as e: + raise Exception(f"Error rendering template - {e}") # don't use verbose_logger.exception, if exception is raised + + +async def _afetch_and_extract_template( + model: str, chat_template: str | None, get_config_fn, get_template_fn +) -> tuple[str, str, str]: + """ + Async version: Fetch template and tokens from HuggingFace. + + Returns: (chat_template, bos_token, eos_token) + """ + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _extract_token_value, + ) + + bos_token = "" + eos_token = "" + + if chat_template is None: + # Fetch or retrieve cached tokenizer config + if model in litellm.known_tokenizer_config: + tokenizer_config = litellm.known_tokenizer_config[model] + else: + tokenizer_config = await get_config_fn(hf_model_name=model) + litellm.known_tokenizer_config.update({model: tokenizer_config}) + + # Try to get chat template from tokenizer_config.json first + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + and "chat_template" in tokenizer_config["tokenizer"] + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + chat_template = tokenizer_data["chat_template"] + else: + # Fallback: Try to fetch chat template from separate .jinja file + template_result: Final = await get_template_fn(hf_model_name=model) + if template_result.get("status") == "success": + chat_template = template_result["chat_template"] + # Still try to get tokens from tokenizer_config if available + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + else: + raise Exception("No chat template found") + + return chat_template, bos_token, eos_token + + +def _fetch_and_extract_template( + model: str, chat_template: str | None, get_config_fn, get_template_fn +) -> tuple[str, str, str]: + """ + Sync version: Fetch template and tokens from HuggingFace. + + Returns: (chat_template, bos_token, eos_token) + """ + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _extract_token_value, + ) + + bos_token = "" + eos_token = "" + + if chat_template is None: + # Fetch or retrieve cached tokenizer config + if model in litellm.known_tokenizer_config: + tokenizer_config = litellm.known_tokenizer_config[model] + else: + tokenizer_config = get_config_fn(hf_model_name=model) + litellm.known_tokenizer_config.update({model: tokenizer_config}) + + # Try to get chat template from tokenizer_config.json first + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + and "chat_template" in tokenizer_config["tokenizer"] + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + chat_template = tokenizer_data["chat_template"] + else: + # Fallback: Try to fetch chat template from separate .jinja file + template_result: Final = get_template_fn(hf_model_name=model) + if template_result.get("status") == "success": + chat_template = template_result["chat_template"] + # Still try to get tokens from tokenizer_config if available + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + else: + raise Exception("No chat template found") + + return chat_template, bos_token, eos_token + + +async def ahf_chat_template(model: str, messages: list, chat_template: str | None = None): + """HuggingFace chat template (async version)""" + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _aget_chat_template_file, + _aget_tokenizer_config, + strftime_now, + ) + + env: Final = ImmutableSandboxedEnvironment() + env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") + env.globals["strftime_now"] = strftime_now + + template, bos_token, eos_token = await _afetch_and_extract_template( + model=model, + chat_template=chat_template, + get_config_fn=_aget_tokenizer_config, + get_template_fn=_aget_chat_template_file, + ) + return _render_chat_template( + env=env, + chat_template=template, + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + ) + + +def hf_chat_template(model: str, messages: list, chat_template: str | None = None): + """HuggingFace chat template (sync version)""" + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _get_chat_template_file, + _get_tokenizer_config, + strftime_now, + ) + + env: Final = ImmutableSandboxedEnvironment() + env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") + env.globals["strftime_now"] = strftime_now + + template, bos_token, eos_token = _fetch_and_extract_template( + model=model, + chat_template=chat_template, + get_config_fn=_get_tokenizer_config, + get_template_fn=_get_chat_template_file, + ) + return _render_chat_template( + env=env, + chat_template=template, + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + ) + + +def deepseek_r1_pt(messages): + return hf_chat_template(model="deepseek-r1/deepseek-r1-7b-instruct", messages=messages) + + +# Anthropic template +def claude_2_1_pt( + messages: list, +): # format - https://docs.anthropic.com/claude/docs/how-to-use-system-prompts + """ + Claude v2.1 allows system prompts (no Human: needed), but requires it be followed by Human: + - you can't just pass a system message + - you can't pass a system message and follow that with an assistant message + if system message is passed in, you can only do system, human, assistant or system, human + + if a system message is passed in and followed by an assistant message, insert a blank human message between them. + + Additionally, you can "put words in Claude's mouth" by ending with an assistant message. + See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth + """ + + class AnthropicConstants(Enum): + HUMAN_PROMPT = "\n\nHuman: " + AI_PROMPT = "\n\nAssistant: " + + prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "user": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{message['content']}" + elif message["role"] == "assistant": + if idx > 0 and messages[idx - 1]["role"] == "system": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}" # Insert a blank human message + prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" + if messages[-1]["role"] != "assistant": + prompt += f"{AnthropicConstants.AI_PROMPT.value}" # prompt must end with \"\n\nAssistant: " turn + return prompt + + +### IBM Granite + + +def ibm_granite_pt(messages: list): + """ + IBM's Granite models uses the template: + <|system|> {system_message} <|user|> {user_message} <|assistant|> {assistant_message} + + See: https://www.ibm.com/docs/en/watsonx-as-a-service?topic=solutions-supported-foundation-models + """ + return custom_prompt( + messages=messages, + role_dict={ + "system": { + "pre_message": "<|system|>\n", + "post_message": "\n", + }, + "user": { + "pre_message": "<|user|>\n", + # Assistant tag is needed in the prompt after the user message + # to avoid the model completing the users sentence before it answers + # https://www.ibm.com/docs/en/watsonx/w-and-w/2.0.x?topic=models-granite-13b-chat-v2-prompting-tips#chat + "post_message": "\n<|assistant|>\n", + }, + "assistant": { + "pre_message": "", + "post_message": "\n", + }, + }, + ).strip() + + +### ANTHROPIC ### + + +def anthropic_pt( + messages: list, +): # format - https://docs.anthropic.com/claude/reference/complete_post + """ + You can "put words in Claude's mouth" by ending with an assistant message. + See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth + """ + + class AnthropicConstants(Enum): + HUMAN_PROMPT = "\n\nHuman: " + AI_PROMPT = "\n\nAssistant: " + + prompt = "" + for idx, message in enumerate(messages): # needs to start with `\n\nHuman: ` and end with `\n\nAssistant: ` + if message["role"] == "user": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + else: + prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" + if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` + prompt = f"{AnthropicConstants.HUMAN_PROMPT.value}" + prompt + if messages[-1]["role"] != "assistant": + prompt += f"{AnthropicConstants.AI_PROMPT.value}" + return prompt + + +def construct_format_parameters_prompt(parameters: dict): + parameter_str = "\n" + for k, v in parameters.items(): + parameter_str += f"<{k}>" + parameter_str += f"{v}" + parameter_str += f"" + parameter_str += "\n" + return parameter_str + + +def construct_format_tool_for_claude_prompt(name, description, parameters): + constructed_prompt: Final = ( + "\n" + f"{name}\n" + "\n" + f"{description}\n" + "\n" + "\n" + f"{construct_format_parameters_prompt(parameters)}\n" + "\n" + "" + ) + return constructed_prompt + + +def construct_tool_use_system_prompt( + tools, +): # from https://github.com/anthropics/anthropic-cookbook/blob/main/function_calling/function_calling.ipynb + tool_str_list: Final = [] + for tool in tools: + tool_function = get_attribute_or_key(tool, "function") + tool_str = construct_format_tool_for_claude_prompt( + get_attribute_or_key(tool_function, "name"), + get_attribute_or_key(tool_function, "description", ""), + get_attribute_or_key(tool_function, "parameters", {}), + ) + tool_str_list.append(tool_str) + tool_use_system_prompt: Final = ( + "In this environment you have access to a set of tools you can use to answer the user's question.\n" + "\n" + "You may call them like this:\n" + "\n" + "\n" + "$TOOL_NAME\n" + "\n" + "<$PARAMETER_NAME>$PARAMETER_VALUE\n" + "...\n" + "\n" + "\n" + "\n" + "\n" + "Here are the tools available:\n" + "\n" + "\n".join([tool_str for tool_str in tool_str_list]) + "\n" + ) + return tool_use_system_prompt + + +def convert_generic_image_chunk_to_openai_image_obj( + image_chunk: GenericImageParsingChunk, +) -> str: + """ + Convert a generic image chunk to an OpenAI image object. + + Input: + GenericImageParsingChunk( + type="base64", + media_type="image/jpeg", + data="...", + ) + + Return: + "data:image/jpeg;base64,{base64_image}" + """ + media_type: Final = image_chunk["media_type"] + return "data:{};{},{}".format(media_type, image_chunk["type"], image_chunk["data"]) + + +def convert_to_anthropic_image_obj(openai_image_url: str, format: str | None) -> GenericImageParsingChunk: + """ + Input: + "image_url": "data:image/jpeg;base64,{base64_image}", + + Return: + "source": { + "type": "base64", + "media_type": "image/jpeg", + "data": {base64_image}, + } + """ + try: + if openai_image_url.startswith("http"): + openai_image_url = convert_url_to_base64(url=openai_image_url) + # Extract the media type and base64 data + media_type, base64_data = openai_image_url.split("data:")[1].split(";base64,") + + if format: + media_type = format + else: + media_type = media_type.replace("\\/", "/") + + return GenericImageParsingChunk( + type="base64", + media_type=media_type, + data=base64_data, + ) + except litellm.ImageFetchError: + raise + except Exception as e: + raise Exception( + f"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{{base64_image}}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']. Error: {e}""" + ) + + +def create_anthropic_image_param( + image_url_input: str | dict, + format: str | None = None, + is_bedrock_invoke: bool = False, +) -> AnthropicMessagesImageParam: + """ + Create an AnthropicMessagesImageParam from an image URL input. + + Supports both URL references (for HTTP/HTTPS URLs) and base64 encoding. + """ + # Extract URL and format from input + if isinstance(image_url_input, str): + image_url = image_url_input + else: + image_url = image_url_input.get("url", "") + if format is None: + format = image_url_input.get("format") + + # Check if the image URL is an HTTP/HTTPS URL + if image_url.startswith("http://") or image_url.startswith("https://"): + # For Bedrock invoke and Vertex AI Anthropic, always convert URLs to base64 + # as these providers don't support URL sources for images + if is_bedrock_invoke or image_url.startswith("http://"): + base64_url: Final = convert_url_to_base64(url=image_url) + image_chunk = convert_to_anthropic_image_obj(openai_image_url=base64_url, format=format) + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + else: + # HTTPS URL - pass directly for regular Anthropic + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSourceUrl( + type="url", + url=image_url, + ), + ) + else: + # Convert to base64 for data URIs or other formats + image_chunk = convert_to_anthropic_image_obj(openai_image_url=image_url, format=format) + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + + +# The following XML functions will be deprecated once JSON schema support is available on Bedrock and Vertex +# ------------------------------------------------------------------------------ +def convert_to_anthropic_tool_result_xml(message: dict) -> str: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + """ + + """ + Anthropic tool_results look like: + + [Successful results] + + + get_current_weather + + function result goes here + + + + + [Error results] + + + error message goes here + + + """ + name: Final = message.get("name") + content = message.get("content", "") + content = content.replace("<", "<").replace(">", ">").replace("&", "&") + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + anthropic_tool_result: Final = ( + "\n" + "\n" + f"{name}\n" + "\n" + f"{content}\n" + "\n" + "\n" + "" + ) + + return anthropic_tool_result + + +def convert_to_anthropic_tool_invoke_xml(tool_calls: list) -> str: + invokes = "" + for tool in tool_calls: + if get_attribute_or_key(tool, "type") != "function": + continue + + tool_function = get_attribute_or_key(tool, "function") + tool_name = get_attribute_or_key(tool_function, "name") + tool_arguments = get_attribute_or_key(tool_function, "arguments") + parsed_args = parse_tool_call_arguments( + tool_arguments, tool_name=tool_name, context="Anthropic XML tool invoke" + ) + if isinstance(parsed_args, dict): + parameters = "".join(f"<{param}>{val}\n" for param, val in parsed_args.items()) + else: + parameters = f"{parsed_args}\n" + invokes += f"\n{tool_name}\n\n{parameters}\n\n" + + anthropic_tool_invoke: Final = f"\n{invokes}" + + return anthropic_tool_invoke + + +def anthropic_messages_pt_xml(messages: list): + """ + format messages for anthropic + 1. Anthropic supports roles like "user" and "assistant", (here litellm translates system-> assistant) + 2. The first message always needs to be of role "user" + 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) + 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) + 5. System messages are a separate param to the Messages API (used for tool calling) + 6. Ensure we only accept role, content. (message.name is not supported) + """ + # add role=tool support to allow function call result/error submission + user_message_types: Final = {"user", "tool"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final = [] + msg_i = 0 + while msg_i < len(messages): + user_content = [] + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "image_url": + format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None + image_param = create_anthropic_image_param(m["image_url"], format=format) + # Convert to dict format for XML version + source = image_param["source"] + if isinstance(source, dict) and source.get("type") == "url": + # Type narrowing for URL source + url_source = cast(AnthropicContentParamSourceUrl, source) + user_content.append( + { + "type": "image", + "source": { + "type": "url", + "url": url_source["url"], + }, + } + ) + else: + # Type narrowing for base64 source + base64_source = cast(AnthropicContentParamSource, source) + user_content.append( + { + "type": "image", + "source": { + "type": "base64", + "media_type": base64_source["media_type"], + "data": base64_source["data"], + }, + } + ) + elif m.get("type", "") == "text": + user_content.append({"type": "text", "text": m["text"]}) + else: + # Tool message content will always be a string + user_content.append( + { + "type": "text", + "text": ( + convert_to_anthropic_tool_result_xml(messages[msg_i]) + if messages[msg_i]["role"] == "tool" + else messages[msg_i]["content"] + ), + } + ) + + msg_i += 1 + + if user_content: + new_messages.append({"role": "user", "content": user_content}) + + assistant_content = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_text = messages[msg_i].get("content") or "" # either string or none + if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion + assistant_text += convert_to_anthropic_tool_invoke_xml(messages[msg_i]["tool_calls"]) + + assistant_content.append({"type": "text", "text": assistant_text}) + msg_i += 1 + + if assistant_content: + new_messages.append({"role": "assistant", "content": assistant_content}) + + if not new_messages or new_messages[0]["role"] != "user": + if litellm.modify_params: + new_messages.insert(0, {"role": "user", "content": [{"type": "text", "text": "."}]}) + else: + raise Exception( + "Invalid first message. Should always start with 'role'='user' for Anthropic. System prompt is sent separately for Anthropic. set 'litellm.modify_params = True' or 'litellm_settings:modify_params = True' on proxy, to insert a placeholder user message - '.' as the first message, " + ) + + if new_messages[-1]["role"] == "assistant": + for content in new_messages[-1]["content"]: + if isinstance(content, dict) and content["type"] == "text": + content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message + + return new_messages + + +# ------------------------------------------------------------------------------ + + +def _azure_tool_call_invoke_helper( + function_call_params: ChatCompletionToolCallFunctionChunk, +) -> ChatCompletionToolCallFunctionChunk | None: + """ + Azure requires 'arguments' to be a string. + """ + if function_call_params.get("arguments") is None: + function_call_params["arguments"] = "" + return function_call_params + + +def _azure_image_url_helper(content: ChatCompletionImageObject): + if isinstance(content["image_url"], str): + content["image_url"] = {"url": content["image_url"]} + else: + content["image_url"] = cast( + ChatCompletionImageUrlObject, + {k: v for k, v in content["image_url"].items() if k != "format"}, + ) + + +def _azure_file_helper(content: ChatCompletionFileObject) -> None: + content["file"] = cast( + ChatCompletionFileObjectFile, + {k: v for k, v in content.get("file", {}).items() if k != "format"}, + ) + + +def convert_to_azure_openai_messages( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: + for m in messages: + if m["role"] == "assistant": + function_call = m.get("function_call", None) + if function_call is not None: + m["function_call"] = _azure_tool_call_invoke_helper(function_call) + + if m["role"] == "user" and isinstance(m.get("content"), list): + for content in m.get("content", []): + if isinstance(content, dict) and content.get("type") == "image_url": + _azure_image_url_helper(cast(ChatCompletionImageObject, content)) + elif isinstance(content, dict) and content.get("type") == "file": + _azure_file_helper(cast(ChatCompletionFileObject, content)) + return messages + + +# ------------------------------------------------------------------------------ + + +def infer_protocol_value( + value: object, +) -> Literal[ + "string_value", + "number_value", + "bool_value", + "struct_value", + "list_value", + "null_value", + "unknown", +]: + if value is None: + return "null_value" + if isinstance(value, int) or isinstance(value, float): + return "number_value" + if isinstance(value, str): + return "string_value" + if isinstance(value, bool): + return "bool_value" + if isinstance(value, dict): + return "struct_value" + if isinstance(value, list): + return "list_value" + + return "unknown" + + +def _gemini_tool_call_invoke_helper( + function_call_params: ChatCompletionToolCallFunctionChunk, + tool_call_id: str | None = None, +) -> VertexFunctionCall | None: + name: Final = function_call_params.get("name", "") or "" + arguments: Final = function_call_params.get("arguments", "") + if ( + isinstance(arguments, str) and len(arguments) == 0 + ): # pass empty dict, if arguments is empty string - prevents call from failing + arguments_dict = { + "type": "object", + } + else: + arguments_dict = json.loads(arguments) + function_call: Final = VertexFunctionCall( + name=name, + args=arguments_dict, + ) + if tool_call_id: + clean_id: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] + if clean_id: + function_call["id"] = clean_id + return function_call + + +def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: str | None) -> str: + """ + Embed thought signature into tool call ID for OpenAI client compatibility. + + Args: + tool_call_id: The tool call ID (e.g., "call_abc123...") + thought_signature: Base64-encoded signature from Gemini response + + Returns: + Tool call ID with embedded signature if present, otherwise original ID + Format: call___thought__ + + See: https://ai.google.dev/gemini-api/docs/thought-signatures + """ + if thought_signature: + return f"{tool_call_id}{THOUGHT_SIGNATURE_SEPARATOR}{thought_signature}" + return tool_call_id + + +def _get_thought_signature_from_tool(tool: dict) -> str | None: + """Extract thought signature from tool call's provider_specific_fields. + + If not provided try to extract thought signature from tool call id + + Checks both tool.provider_specific_fields and tool.function.provider_specific_fields. + Returns None when the tool call carries no signature; callers decide whether a + placeholder signature is needed. + """ + # First check tool's provider_specific_fields + provider_fields: Final = tool.get("provider_specific_fields") or {} + if isinstance(provider_fields, dict): + signature = provider_fields.get("thought_signature") + if signature: + return signature + + # Then check function's provider_specific_fields + function: Final = tool.get("function") + if function: + if isinstance(function, dict): + func_provider_fields: Final = function.get("provider_specific_fields") or {} + if isinstance(func_provider_fields, dict): + signature = func_provider_fields.get("thought_signature") + if signature: + return signature + elif hasattr(function, "provider_specific_fields") and function.provider_specific_fields: + if isinstance(function.provider_specific_fields, dict): + signature = function.provider_specific_fields.get("thought_signature") + if signature: + return signature + # Check if thought signature is embedded in tool call ID + tool_call_id: Final = tool.get("id") + if tool_call_id and THOUGHT_SIGNATURE_SEPARATOR in tool_call_id: + parts: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1) + if len(parts) == 2: + _, signature = parts + return signature + return None + + +def _get_dummy_thought_signature() -> str: + """Generate a dummy thought signature for models that require it. + + This is used when transferring conversation history from older models + (like gemini-2.5-flash) to gemini-3, which requires thought_signature + for strict validation. Google documents it as a last resort that "will + negatively impact model performance", so callers must only fall back to it + when no real signature is available. + + See: + https://ai.google.dev/gemini-api/docs/thought-signatures#faqs + https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures + """ + dummy_data: Final = b"skip_thought_signature_validator" + return base64.b64encode(dummy_data).decode("utf-8") + + +def convert_to_gemini_tool_call_invoke( + message: ChatCompletionAssistantMessage, + model: str | None = None, + forward_function_call_id: bool = False, +) -> list[VertexPartType]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + """ + Gemini tool call invokes: + { + "role": "model", + "parts": [ + { + "functionCall": { + "name": "get_current_weather", + "args": { + "unit": "fahrenheit", + "predicted_temperature": 45, + "location": "Boston, MA", + } + } + } + ] + } + """ + + """ + - json.load the arguments + """ + try: + _parts_list: Final[list[VertexPartType]] = [] + tool_calls: Final = message.get("tool_calls", None) + function_call: Final = message.get("function_call", None) + + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, + ) + + needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model) + + if tool_calls is not None: + for tool in tool_calls: + if "function" in tool: + gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper( + function_call_params=tool["function"], + tool_call_id=(tool.get("id") if forward_function_call_id else None), + ) + if gemini_function_call is not None: + part_dict: VertexPartType = {"function_call": gemini_function_call} + thought_signature = _get_thought_signature_from_tool(dict(tool)) + # Gemini signs only the first functionCall part of a parallel batch, so scope the + # placeholder fallback to that part instead of fabricating one per sibling call: + # https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example + is_first_function_call = len(_parts_list) == 0 + if not thought_signature and is_first_function_call and needs_dummy_signature: + thought_signature = _get_dummy_thought_signature() + if thought_signature: + part_dict["thoughtSignature"] = thought_signature + + _parts_list.append(part_dict) + else: # don't silently drop params. Make it clear to user what's happening. + raise Exception( + f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {tool}" + ) + elif function_call is not None: + gemini_function_call = _gemini_tool_call_invoke_helper(function_call_params=function_call) + if gemini_function_call is not None: + part_dict_function: Final[VertexPartType] = {"function_call": gemini_function_call} + + # Extract thought signature from function_call's provider_specific_fields + thought_signature = None + provider_fields: Final = ( + function_call.get("provider_specific_fields") if isinstance(function_call, dict) else {} + ) + if isinstance(provider_fields, dict): + thought_signature = provider_fields.get("thought_signature") + + # If no signature found and model is gemini-3, use dummy signature + if not thought_signature and needs_dummy_signature: + thought_signature = _get_dummy_thought_signature() + + if thought_signature: + part_dict_function["thoughtSignature"] = thought_signature + + _parts_list.append(part_dict_function) + else: # don't silently drop params. Make it clear to user what's happening. + raise Exception( + f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {message}" + ) + return _parts_list + except Exception as e: + raise Exception(f"Unable to convert openai tool calls={message} to gemini tool calls. Received error={e}") + + +def convert_to_gemini_tool_call_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + last_message_with_tool_calls: dict | None, + forward_function_call_id: bool = False, +) -> VertexPartType | list[VertexPartType]: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "content": "function result goes here", + }, + + # NOTE: Function messages have been deprecated + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + + Supports content with images for Computer Use: + { + "role": "tool", + "tool_call_id": "call_abc123", + "content": [ + {"type": "text", "text": "I found the requested image:"}, + {"type": "input_image", "image_url": "https://example.com/image.jpg" } + ] + } + """ + from litellm.types.llms.vertex_ai import BlobType + + content_str: str = "" + inline_data_list: Final[list[BlobType]] = [] + + if "content" in message: + if isinstance(message["content"], str): + content_str = message["content"] + # Detect data-URL images (e.g. from Anthropic tool_result with a single image block + # that was serialised as a plain string by translate_anthropic_messages_to_openai) + # and promote them to inline_data so Gemini receives actual image bytes. + if content_str[:5].lower() == "data:" and ";base64," in content_str: + try: + mime_rest: Final = content_str[5:].split(";base64,", 1) + if len(mime_rest) == 2 and mime_rest[0].startswith("image/"): + # Strip any extra parameters (e.g. ";charset=UTF-8") from the MIME segment + clean_mime: Final = mime_rest[0].split(";")[0].strip() + inline_data_list.append(BlobType(data=mime_rest[1], mime_type=clean_mime)) + content_str = "" + except Exception as e: + verbose_logger.warning("Failed to parse data URL in tool response: %s", e) + elif isinstance(message["content"], list): + content_list: Final = message["content"] + for content in content_list: + content_type = content.get("type", "") + if content_type == "text": + content_str += content.get("text", "") + elif content_type == "image": # pyright: ignore[reportUnnecessaryComparison] # loose runtime dict + # Anthropic-native image block: {"type": "image", "source": {"type": "base64", ...}} + source = content.get("source", {}) + if isinstance(source, dict) and source.get("type") == "base64": + try: + inline_data_list.append( + BlobType( + data=source.get("data", ""), + mime_type=source.get("media_type", "image/jpeg"), + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process Anthropic image block in tool response: %s", e) + elif content_type in ("input_image", "image_url"): + # Extract image for inline_data (for Computer Use screenshots and tool results) + image_url_data = content.get("image_url", "") + image_url = image_url_data.get("url", "") if isinstance(image_url_data, dict) else image_url_data + + if image_url: + # Convert image to base64 blob format for Gemini + try: + image_obj = convert_to_anthropic_image_obj(image_url, format=None) + inline_data_list.append( + BlobType( + data=image_obj["data"], + mime_type=image_obj["media_type"], + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process image in tool response: %s", e) + elif content_type in ("file", "input_file"): # pyright: ignore[reportUnnecessaryContains] # loose runtime dict + # Extract file for inline_data (for tool results with PDF, audio, video, etc.) + file_data = content.get("file_data", "") + if not file_data: + file_content = content.get("file", {}) + file_data = ( + file_content.get("file_data", "") + if isinstance(file_content, dict) + else file_content + if isinstance(file_content, str) + else "" + ) + + if file_data: + # Convert file to base64 blob format for Gemini + try: + file_obj = convert_to_anthropic_image_obj(file_data, format=None) + inline_data_list.append( + BlobType( + data=file_obj["data"], + mime_type=file_obj["media_type"], + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process file in tool response: %s", e) + name: str | None = message.get("name", "") + + # Recover name from last message with tool calls + if last_message_with_tool_calls: + tools: Final = last_message_with_tool_calls.get("tool_calls", []) + msg_tool_call_id: Final = message.get("tool_call_id", None) + for tool in tools: + prev_tool_call_id = tool.get("id", None) + if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: + name = tool.get("function", {}).get("name", "") + + # Echo the OpenAI tool_call_id on functionResponse (strip thought-signature suffix). + gemini_call_id: str | None = None + if forward_function_call_id: + raw_tool_call_id: Final = message.get("tool_call_id") + if raw_tool_call_id and isinstance(raw_tool_call_id, str): + stripped_id: Final = raw_tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] + if stripped_id: + gemini_call_id = stripped_id + + if not name: + raise Exception( + f"Missing corresponding tool call for tool response message. Received - message={message}, last_message_with_tool_calls={last_message_with_tool_calls}" + ) + + # Parse response data - support both JSON string and plain string + # For Computer Use, the response should contain structured data like {"url": "..."} + response_data: dict + try: + if content_str.strip().startswith("{") or content_str.strip().startswith("["): + # Try to parse as JSON (for Computer Use structured responses) + parsed: Final = json.loads(content_str) + if isinstance(parsed, dict): + response_data = parsed # Use the parsed JSON directly + else: + response_data = {"content": content_str} + else: + response_data = {"content": content_str} + except (json.JSONDecodeError, ValueError): + # Not valid JSON, wrap in content field + response_data = {"content": content_str} + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + _function_response: Final = VertexFunctionResponse( + name=name, + response=response_data, + ) + if gemini_call_id: + _function_response["id"] = gemini_call_id + + _part: Final[VertexPartType] = {"function_response": _function_response} + + # For multimodal function responses, Gemini expects media parts nested + # inside functionResponse.parts instead of sibling content parts. + if inline_data_list: + _function_response["parts"] = [{"inline_data": inline_data} for inline_data in inline_data_list] + return [_part] + + return _part + + +_TOOL_USE_ID_FALLBACK: Final = "tool_use_id" +_ANTHROPIC_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_-]") +_BEDROCK_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_.:-]") +_BEDROCK_TOOL_USE_ID_MAX_LEN: Final = 64 +_BEDROCK_TOOL_USE_ID_HASH_LEN: Final = 8 + + +def _replace_invalid_tool_use_id_chars(tool_use_id: str, invalid_chars: re.Pattern[str]) -> str: + return invalid_chars.sub("_", tool_use_id) or _TOOL_USE_ID_FALLBACK + + +def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str: + """Anthropic requires tool_use_id to match ^[a-zA-Z0-9_-]+$.""" + return _replace_invalid_tool_use_id_chars(tool_use_id, _ANTHROPIC_TOOL_USE_ID_INVALID_CHARS) + + +def _sanitize_bedrock_tool_use_id(tool_use_id: str) -> str: + """ + Bedrock Converse requires toolUseId to match [a-zA-Z0-9_.:-]+ and be at most 64 chars. + Ids that need rewriting get a short hash of the original appended so two ids that only + differ in a replaced char or past the cut still map to distinct values. + """ + sanitized: Final = _replace_invalid_tool_use_id_chars(tool_use_id, _BEDROCK_TOOL_USE_ID_INVALID_CHARS) + if sanitized == tool_use_id and len(sanitized) <= _BEDROCK_TOOL_USE_ID_MAX_LEN: + return sanitized + digest: Final = hashlib.sha256(tool_use_id.encode()).hexdigest()[:_BEDROCK_TOOL_USE_ID_HASH_LEN] + return f"{sanitized[: _BEDROCK_TOOL_USE_ID_MAX_LEN - _BEDROCK_TOOL_USE_ID_HASH_LEN - 1]}_{digest}" + + +_ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES: Final = {"application/pdf", "text/plain"} + + +def _is_anthropic_document_data_uri(url: str) -> bool: + # Anthropic's base64 document source accepts only application/pdf and + # text/plain (see select_anthropic_content_block_type_for_file). Routing + # other mimes here would produce a document block the API rejects, so we + # leave them on the image code path. + match: Final = re.match(r"data:([^;,]+)", url) + if not match: + return False + return match.group(1) in _ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES + + +def convert_to_anthropic_tool_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + force_base64: bool = False, +) -> AnthropicMessagesToolResultParam: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + + """ + Anthropic tool_results look like: + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_01A09q90qw90lq917835lq9", + "content": "ConnectionError: the weather service API is not available (HTTP 500)", + # "is_error": true + } + ] + } + """ + anthropic_content: ( + str + | list[ + AnthropicMessagesToolResultContent + | AnthropicMessagesImageParam + | AnthropicMessagesDocumentParam + | ToolReference + ] + ) = "" + if isinstance(message["content"], str): + anthropic_content = message["content"] + elif isinstance(message["content"], list): + content_list: Final = message["content"] + anthropic_content_list: list[ + AnthropicMessagesToolResultContent + | AnthropicMessagesImageParam + | AnthropicMessagesDocumentParam + | ToolReference + ] = [] + for content in content_list: + if content["type"] == "text": + # Only include cache_control if explicitly set and not None + # to avoid sending "cache_control": null which breaks some API channels + text_content: AnthropicMessagesToolResultContent = { + "type": "text", + "text": content["text"], + } + cache_control_value = content.get("cache_control") + if cache_control_value is not None: + text_content["cache_control"] = cache_control_value + anthropic_content_list.append(text_content) + elif content["type"] == "image_url": + image_url_value = content["image_url"] + format = image_url_value.get("format") if isinstance(image_url_value, dict) else None + url_str = image_url_value.get("url") if isinstance(image_url_value, dict) else image_url_value + # Data URIs with non-image mime types (e.g. application/pdf) must + # translate to Anthropic document blocks, not image blocks — + # wrapping a PDF in `type: "image"` is rejected by the API. + if isinstance(url_str, str) and _is_anthropic_document_data_uri(url_str): + synth_file_message: ChatCompletionFileObject = { + "type": "file", + "file": {"file_data": url_str}, + } + _document_block = anthropic_process_openai_file_message(synth_file_message) + _document_block = add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, _document_block), + original_content_element=content, + ) + anthropic_content_list.append(cast(AnthropicMessagesDocumentParam, _document_block)) + else: + _anthropic_image_param = create_anthropic_image_param( + image_url_value, + format=format, + is_bedrock_invoke=force_base64, + ) + _anthropic_image_param = add_cache_control_to_content( + anthropic_content_element=_anthropic_image_param, + original_content_element=content, + ) + anthropic_content_list.append(cast(AnthropicMessagesImageParam, _anthropic_image_param)) + elif content["type"] == "tool_reference": + anthropic_content_list.append(ToolReference(type="tool_reference", tool_name=content["tool_name"])) + elif content["type"] == "file": + file_content = cast(ChatCompletionFileObject, content) + _file_block = anthropic_process_openai_file_message(file_content) + _file_block = add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, _file_block), + original_content_element=content, + ) + anthropic_content_list.append(_file_block) + + anthropic_content = anthropic_content_list + anthropic_tool_result: AnthropicMessagesToolResultParam | None = None + ## PROMPT CACHING CHECK ## + cache_control: Final = message.get("cache_control", None) + if message["role"] == "tool": + tool_message: Final[ChatCompletionToolMessage] = message + tool_call_id: str = tool_message["tool_call_id"] + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + anthropic_tool_result = AnthropicMessagesToolResultParam( + type="tool_result", + tool_use_id=sanitized_tool_use_id, + content=anthropic_content, + ) + + if message["role"] == "function": + function_message: Final[ChatCompletionFunctionMessage] = message + tool_call_id = function_message.get("tool_call_id") or str(uuid.uuid4()) + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) + anthropic_tool_result = AnthropicMessagesToolResultParam( + type="tool_result", + tool_use_id=sanitized_tool_use_id, + content=anthropic_content, + ) + + if anthropic_tool_result is None: + raise Exception(f"Unable to parse anthropic tool result for message: {message}") + if cache_control is not None: + anthropic_tool_result["cache_control"] = cache_control + return anthropic_tool_result + + +def convert_function_to_anthropic_tool_invoke( + function_call: dict | ChatCompletionToolCallFunctionChunk, +) -> list[AnthropicMessagesToolUseParam]: + try: + _name: Final = get_attribute_or_key(function_call, "name") or "" + _arguments: Final = get_attribute_or_key(function_call, "arguments") + + tool_input: Final = parse_tool_call_arguments( + _arguments, tool_name=_name, context="Anthropic function to tool invoke" + ) + + anthropic_tool_invoke: Final = [ + AnthropicMessagesToolUseParam( + type="tool_use", + id=str(uuid.uuid4()), + name=_name, + input=tool_input, + ) + ] + return anthropic_tool_invoke + except Exception as e: + raise e + + +def _find_server_tool_result( + tool_id: str, + web_search_results: Sequence[object] | None, + tool_results: Sequence[object] | None, +) -> dict[str, object] | None: + candidates: Final = (*(web_search_results or ()), *(tool_results or ())) + return next( + (result for result in candidates if isinstance(result, dict) and result.get("tool_use_id") == tool_id), + None, + ) + + +def convert_to_anthropic_tool_invoke( + tool_calls: list[ChatCompletionAssistantToolCall], + web_search_results: Sequence[object] | None = None, + tool_results: Sequence[object] | None = None, +) -> list[AnthropicMessagesToolUseParam | dict[str, Any]]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + + """ + Anthropic tool invokes: + { + "role": "assistant", + "content": [ + { + "type": "text", + "text": "To answer this question, I will: 1. Use the get_weather tool to get the current weather in San Francisco. 2. Use the get_time tool to get the current time in the America/Los_Angeles timezone, which covers San Francisco, CA." + }, + { + "type": "tool_use", + "id": "toolu_01A09q90qw90lq917835lq9", + "name": "get_weather", + "input": {"location": "San Francisco, CA"} + } + ] + } + + For server-side tools (web_search), we need to reconstruct: + - server_tool_use blocks (id starts with "srvtoolu_") + - web_search_tool_result blocks (from provider_specific_fields) + + Fixes: https://github.com/BerriAI/litellm/issues/17737 + """ + anthropic_tool_invoke: Final[list[AnthropicMessagesToolUseParam | dict[str, object]]] = [] + + for tool in tool_calls: + if get_attribute_or_key(tool, "type") != "function": + continue + + tool_id = cast(str, get_attribute_or_key(tool, "id")) + tool_name = cast( + str, + get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), + ) + tool_input = parse_tool_call_arguments( + get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments"), + tool_name=tool_name, + context="Anthropic tool invoke", + ) + + server_tool_result = ( + _find_server_tool_result(tool_id, web_search_results, tool_results) + if tool_id.startswith("srvtoolu_") + else None + ) + if server_tool_result is not None: + anthropic_tool_invoke.append( + { + "type": "server_tool_use", + "id": tool_id, + "name": tool_name, + "input": tool_input, + } + ) + anthropic_tool_invoke.append(server_tool_result) + else: + sanitized_tool_id = _sanitize_anthropic_tool_use_id(tool_id) + _anthropic_tool_use_param = AnthropicMessagesToolUseParam( + type="tool_use", + id=sanitized_tool_id, + name=tool_name, + input=tool_input, + ) + + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_tool_use_param, + original_content_element=dict(tool), + ) + + if "cache_control" in _content_element: + _anthropic_tool_use_param["cache_control"] = _content_element["cache_control"] + + anthropic_tool_invoke.append(_anthropic_tool_use_param) + + return anthropic_tool_invoke + + +def add_cache_control_to_content( + anthropic_content_element: dict + | AnthropicMessagesImageParam + | AnthropicMessagesTextParam + | AnthropicMessagesDocumentParam + | AnthropicMessagesToolUseParam + | ChatCompletionThinkingBlock, + original_content_element: dict | AllMessageValues, +): + cache_control_param: Final = original_content_element.get("cache_control") + if cache_control_param is not None and isinstance(cache_control_param, dict): + transformed_param: Final = ChatCompletionCachedContent(**cache_control_param) + + anthropic_content_element["cache_control"] = transformed_param + + return anthropic_content_element + + +def _anthropic_content_element_factory( + image_chunk: GenericImageParsingChunk, +) -> AnthropicMessagesImageParam | AnthropicMessagesDocumentParam: + if image_chunk["media_type"] == "application/pdf": + _anthropic_content_element: AnthropicMessagesDocumentParam | AnthropicMessagesImageParam = ( + AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + ) + else: + _anthropic_content_element = AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + + return _anthropic_content_element + + +def select_anthropic_content_block_type_for_file( + format: str, +) -> Literal["document", "image", "container_upload"]: + if format == "application/pdf" or format == "text/plain": + return "document" + elif format in ["image/jpeg", "image/png", "image/gif", "image/webp"]: + return "image" + else: + return "container_upload" + + +def anthropic_infer_file_id_content_type( + file_id: str, +) -> Literal["document_url", "container_upload"]: + """ + Use when 'format' not provided. + + - URL's - assume are document_url + - Else - assume is container_upload + """ + if file_id.startswith("http") or file_id.startswith("https"): + return "document_url" + else: + return "container_upload" + + +def anthropic_process_openai_file_message( + message: ChatCompletionFileObject, +) -> AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam: + file_message: Final = cast(ChatCompletionFileObject, message) + file_sub: Final = file_message.get("file") + if file_sub is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="anthropic", + ) + file_data: Final = file_sub.get("file_data") + file_id: Final = file_sub.get("file_id") + format: Final = file_sub.get("format") + if file_data: + image_chunk: Final = convert_to_anthropic_image_obj( + openai_image_url=file_data, + format=format, + ) + anthropic_document_param: Final = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + return anthropic_document_param + elif file_id: + content_block_type: Final = ( + select_anthropic_content_block_type_for_file(format) + if format + else anthropic_infer_file_id_content_type(file_id) + ) + return_block_param: ( + AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam | None + ) = None + if content_block_type == "document": + return_block_param = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSourceFileId( + type="file", + file_id=file_id, + ), + ) + elif content_block_type == "document_url": + return_block_param = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSourceUrl( + type="url", + url=file_id, + ), + ) + elif content_block_type == "image": + return_block_param = AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSourceFileId( + type="file", + file_id=file_id, + ), + ) + elif content_block_type == "container_upload": + return_block_param = AnthropicMessagesContainerUploadParam(type="container_upload", file_id=file_id) + + if return_block_param is None: + raise Exception(f"Unable to parse anthropic file message: {message}") + return return_block_param + raise Exception(f"Either file_data or file_id must be present in the file message: {message}") + + +_EMPTY_TEXT_PLACEHOLDER: Final = "[System: Empty message content sanitised to satisfy protocol]" + + +def _sanitize_empty_text_content( + message: AllMessageValues, +) -> AllMessageValues: + """ + Case C: Sanitize empty text content + - Replace empty or whitespace-only text content with a placeholder message. + - Handles both string content and list-of-blocks content (rewriting only + the empty text blocks in place; non-text blocks like images are left + untouched). + + Returns: + The message with sanitized content if needed, otherwise the original message + """ + if message.get("role") not in ["user", "assistant"]: + return message + + content: Final = message.get("content") + + if isinstance(content, str): + if not content or not content.strip(): + message = cast(AllMessageValues, dict(message)) # Make a copy + message["content"] = _EMPTY_TEXT_PLACEHOLDER + verbose_logger.debug( + "_sanitize_empty_text_content: Replaced empty text content in %s message", message.get("role") + ) + return message + + if isinstance(content, list): + # Walk the blocks and rewrite any empty text blocks. We rewrite (rather + # than drop) so callers don't end up with an entirely empty content + # list, which Anthropic also rejects. + new_blocks: Final[list[Any]] = [] + rewrote_any = False + for block in content: + if isinstance(block, dict) and block.get("type") == "text": + text = block.get("text") + if not isinstance(text, str) or not text or not text.strip(): + new_block = dict(block) + new_block["text"] = _EMPTY_TEXT_PLACEHOLDER + new_blocks.append(new_block) + rewrote_any = True + continue + new_blocks.append(block) + + if rewrote_any: + message = cast(AllMessageValues, dict(message)) # Make a copy + message["content"] = new_blocks + verbose_logger.debug( + "_sanitize_empty_text_content: Replaced empty text block(s) in %s message", message.get("role") + ) + + return message + + +def _add_missing_tool_results( + current_message: AllMessageValues, + messages: list[AllMessageValues], + current_index: int, +) -> tuple[list[AllMessageValues], int]: + """ + Case A: Missing tool_result for tool_use (orphaned tool calls) + - If an assistant message has tool_calls but no corresponding tool result follows, + add a dummy tool result message indicating the user did not provide the result. + + Returns: + A tuple of: + - List containing the assistant message, followed by existing tool results, + followed by any dummy tool results needed + - Number of original messages consumed (to adjust iteration index) + """ + result_messages: Final[list[AllMessageValues]] = [] + tool_calls: Final = current_message.get("tool_calls") + + if not tool_calls or len(cast(list, tool_calls)) == 0: + return ([current_message], 0) + + # Collect all tool_call_ids from this assistant message + expected_tool_call_ids: Final = set() + for tool_call in cast(list, tool_calls): + tool_call_id = None + if isinstance(tool_call, dict): + tool_call_id = tool_call.get("id") + else: + tool_call_id = getattr(tool_call, "id", None) + if tool_call_id: + expected_tool_call_ids.add(tool_call_id) + + # Collect actual tool result messages that follow this assistant message + found_tool_call_ids: Final = set() + actual_tool_results: Final[list[AllMessageValues]] = [] + j = current_index + 1 + + while j < len(messages): + next_msg = messages[j] + next_role = next_msg.get("role") + + if next_role == "assistant": + break + + if next_role in ["tool", "function"]: + tool_call_id = next_msg.get("tool_call_id") + if tool_call_id and tool_call_id in expected_tool_call_ids: + found_tool_call_ids.add(tool_call_id) + actual_tool_results.append(next_msg) + + j += 1 + + # Find missing tool results + missing_tool_call_ids: Final = expected_tool_call_ids - found_tool_call_ids + + if missing_tool_call_ids: + verbose_logger.debug( + "_add_missing_tool_results: Found %s orphaned tool calls. Adding dummy tool results.", + len(missing_tool_call_ids), + ) + + result_messages.append(current_message) + + # Add existing tool results FIRST + result_messages.extend(actual_tool_results) + + # Then add dummy tool results for missing ones + for tool_call_id in missing_tool_call_ids: + tool_name = "unknown_tool" + for tool_call in cast(list, tool_calls): + tc_id = None + if isinstance(tool_call, dict): + tc_id = tool_call.get("id") + else: + tc_id = getattr(tool_call, "id", None) + + if tc_id == tool_call_id: + if isinstance(tool_call, dict): + function = tool_call.get("function", {}) + if isinstance(function, dict): + tool_name = function.get("name", "unknown_tool") + else: + tool_name = getattr(function, "name", "unknown_tool") + else: + function = getattr(tool_call, "function", None) + if function: + tool_name = getattr(function, "name", "unknown_tool") + break + + dummy_tool_result: ChatCompletionToolMessage = { + "role": "tool", + "tool_call_id": tool_call_id, + "content": f"[System: Tool execution skipped/interrupted by user. No result provided for tool '{tool_name}'.]", + } + result_messages.append(dummy_tool_result) + + # Return the messages and the number of original messages to skip + return (result_messages, len(actual_tool_results)) + + return ([current_message], 0) + + +def _is_orphaned_tool_result( + current_message: AllMessageValues, + sanitized_messages: list[AllMessageValues], +) -> bool: + """ + Case B: Orphaned tool_result (unexpected result) + - Check if a tool message references a tool_call_id that doesn't exist in the previous + assistant message. + + Returns: + True if this is an orphaned tool result that should be removed, False otherwise + """ + if current_message.get("role") not in ["tool", "function"]: + return False + + tool_call_id: Final = current_message.get("tool_call_id") + + if not tool_call_id: + return False + + # Look back to find the most recent assistant message with tool_calls + found_matching_tool_call = False + + for j in range(len(sanitized_messages) - 1, -1, -1): + prev_msg = sanitized_messages[j] + if prev_msg.get("role") == "assistant": + tool_calls = prev_msg.get("tool_calls") + if tool_calls: + for tool_call in cast(list, tool_calls): + tc_id = None + if isinstance(tool_call, dict): + tc_id = tool_call.get("id") + else: + tc_id = getattr(tool_call, "id", None) + + if tc_id == tool_call_id: + found_matching_tool_call = True + break + + break + + if not found_matching_tool_call: + verbose_logger.debug("_is_orphaned_tool_result: Found orphaned tool result with redacted tool_call_id") + return True + + return False + + +def _declared_tool_call_ids(message: Mapping[str, object]) -> frozenset[str]: + tool_calls: Final = message.get("tool_calls") + if not isinstance(tool_calls, list): + return frozenset() + return frozenset( + str(tool_call["id"]) for tool_call in tool_calls if isinstance(tool_call, Mapping) and tool_call.get("id") + ) + + +def group_tool_exchanges(messages: Sequence[Mapping[str, object]]) -> tuple[tuple[int, ...], ...]: + """Group message indices into tool exchanges: an assistant row that made + tool calls, together with the tool rows answering the ids it declared. + + Membership is by ``tool_call_id`` ownership rather than adjacency, so a tool + row belonging to some other call opens its own group instead of being swept + into the exchange it happens to sit next to. Every other row is its own + group. Groups stay contiguous and in order, so a caller can convert or + protect them without reordering the conversation. + + Callers need this because an assistant row and the tool rows answering it + are only well-formed together: ``sanitize_messages_for_tool_calling`` reads + an assistant row whose results are missing as an orphaned tool call, and + a tool row whose call is missing as an orphaned result. + """ + return tuple(_iter_tool_exchange_groups(messages)) + + +def _iter_tool_exchange_groups(messages: Sequence[Mapping[str, object]]) -> Iterator[tuple[int, ...]]: + index = 0 + while index < len(messages): + declared = _declared_tool_call_ids(messages[index]) + end = index + 1 + while ( + declared + and end < len(messages) + and messages[end].get("role") in ("tool", "function") + and str(messages[end].get("tool_call_id")) in declared + ): + end += 1 + yield tuple(range(index, end)) + index = end + + +def sanitize_messages_for_tool_calling( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: + """ + Sanitize messages for tool calling to handle common issues when modify_params=True: + + Case A: Missing tool_result for tool_use (orphaned tool calls) + - If an assistant message has tool_calls but no corresponding tool result follows, + add a dummy tool result message indicating the user did not provide the result. + + Case B: Orphaned tool_result (unexpected result) + - If a tool message references a tool_call_id that doesn't exist in the previous + assistant message, remove that tool message. + + Case C: Empty text content + - Replace empty or whitespace-only text content with a placeholder message. + + Case D: Duplicate tool_result for same tool_use (duplicate results) + - If multiple tool messages reference the same tool_call_id, keep only the last + occurrence. Anthropic requires exactly one tool_result per tool_use and rejects + with: "each tool_use must have a single result". + + This function operates on OpenAI format messages before they are converted to + provider-specific formats. + """ + if not litellm.modify_params: + return messages + + sanitized_messages: list[AllMessageValues] = [] + i = 0 + + while i < len(messages): + current_message = messages[i] + + # Case C: Sanitize empty text content + current_message = _sanitize_empty_text_content(current_message) + + # Case A: Check if assistant message has tool_calls without following tool results + if current_message.get("role") == "assistant": + result_messages, messages_consumed = _add_missing_tool_results(current_message, messages, i) + + # If dummy tool results were added, extend sanitized_messages and skip consumed messages + if len(result_messages) > 1: + sanitized_messages.extend(result_messages) + # Skip the assistant message and any actual tool results that were included + i += 1 + messages_consumed + continue + + # Case B: Check for orphaned tool results + if _is_orphaned_tool_result(current_message, sanitized_messages): + i += 1 + continue # Skip this orphaned tool result + + # Add the message to sanitized list + sanitized_messages.append(current_message) + i += 1 + + # Case D: Deduplicate tool results with the same tool_call_id. + # Anthropic requires exactly one tool_result per tool_use. Session history + # (e.g. from conversation resume) can contain duplicate tool_result messages + # for the same tool_call_id. Keep only the last occurrence *within each + # contiguous block of tool results following an assistant message*. This + # avoids dropping results from earlier turns if a tool_call_id is reused. + # + # NOTE: This intentionally keeps the *last* occurrence (most complete for + # session-resume duplicates), unlike _deduplicate_bedrock_content_blocks + # which keeps the *first*. The Bedrock case handles provider-side content + # block duplication where the first is authoritative; here the duplicate + # arises from history replay where the last entry is the final state. + duplicates_to_remove: Final[set[int]] = set() + seen_in_block: dict[str, int] = {} # tool_call_id -> index (reset per block) + for idx, msg in enumerate(sanitized_messages): + role = msg.get("role") + tcid = msg.get("tool_call_id") if role in ["tool", "function"] else None + if tcid and isinstance(tcid, str): + if tcid in seen_in_block: + # Mark the earlier occurrence for removal (keep latest) + duplicates_to_remove.add(seen_in_block[tcid]) + verbose_logger.warning( + "sanitize_messages_for_tool_calling: dropping duplicate " + "tool_result with tool_call_id=%s. This may indicate " + "duplicate tool messages in conversation history.", + tcid, + ) + seen_in_block[tcid] = idx + elif role not in ("tool", "function"): + # Non-tool message (user, assistant, system) marks a + # conversational-turn boundary — reset tracking. + # Tool/function messages with no tool_call_id are malformed; + # they should NOT reset the block because they don't represent + # a turn boundary and would mask real within-block duplicates. + seen_in_block = {} + + if duplicates_to_remove: + sanitized_messages = [msg for idx, msg in enumerate(sanitized_messages) if idx not in duplicates_to_remove] + + return sanitized_messages + + +def _drop_unsignable_thinking_blocks( + thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], +) -> list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock]: + return [block for block in thinking_blocks if not is_unsignable_thinking_block(block)] + + +_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues] + + +def anthropic_messages_pt( + messages: list[AllMessageValues], + model: str, + llm_provider: str, +) -> _AnthropicMessageList: + """ + format messages for anthropic + 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately). + Models flagged ``supports_mid_conversation_system`` also accept "system" inside + messages after a user turn; the caller decides placement, this keeps such messages. + 2. The first message always needs to be of role "user" + 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) + 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) + 5. System messages are a separate param to the Messages API + 6. Ensure we only accept role, content. (message.name is not supported) + """ + # Sanitize messages for tool calling issues when modify_params=True + messages = sanitize_messages_for_tool_calling(messages) + + # Anthropic rejects empty text content blocks with: + # "messages: text content blocks must be non-empty" + # OpenAI/other providers silently tolerate `{"role": "user", "content": ""}`, + # so callers (and upstream agent frameworks like pydantic-ai) routinely + # send empty user/assistant turns. We always rewrite these to a placeholder + # for Anthropic-shaped requests, independent of `litellm.modify_params`, + # because there is no way to "pass through" an empty text block — the + # request will always 400 otherwise. The richer tool-call sanitization + # (Cases A/B/D in `sanitize_messages_for_tool_calling`) remains gated on + # `modify_params` because it actually mutates conversation structure. + messages = [_sanitize_empty_text_content(m) for m in messages] + + # add role=tool support to allow function call result/error submission + user_message_types: Final = {"user", "tool", "function"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract + + if len(messages) == 0: + if not litellm.modify_params: + raise litellm.BadRequestError( + message=f"Anthropic requires at least one non-system message. Either provide one, or set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add the dummy user message - {DEFAULT_USER_CONTINUE_MESSAGE_TYPED}.", + model=model, + llm_provider=llm_provider, + ) + else: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE_TYPED) + + # Bedrock invoke models have format: invoke/... + # Vertex AI Anthropic also doesn't support URL sources for images + is_bedrock_invoke = model.lower().startswith("invoke/") + is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False + force_base64 = is_bedrock_invoke or is_vertex_ai + + msg_i = 0 + while msg_i < len(messages): + user_content: list[AnthropicMessagesUserMessageValues] = [] + init_msg_i = msg_i + if isinstance(messages[msg_i], BaseModel): + messages[msg_i] = dict(messages[msg_i]) + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + user_message_types_block: ( + ChatCompletionToolMessage | ChatCompletionUserMessage | ChatCompletionFunctionMessage + ) = messages[msg_i] + if user_message_types_block["role"] == "user": + if isinstance(user_message_types_block["content"], list): + user_content_start: int = len(user_content) + for m in user_message_types_block["content"]: + if m.get("type", "") == "image_url": + m = cast(ChatCompletionImageObject, m) + format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None + # Convert ChatCompletionImageUrlObject to dict if needed + image_url_value = m["image_url"] + if isinstance(image_url_value, str): + image_url_input: str | dict[str, object] = image_url_value + else: + # ChatCompletionImageUrlObject or dict case - convert to dict + image_url_input = { + "url": image_url_value["url"], + "format": image_url_value.get("format"), + } + # Bedrock invoke models have format: invoke/... + # Vertex AI Anthropic also doesn't support URL sources for images + is_bedrock_invoke = model.lower().startswith("invoke/") + is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False + force_base64 = is_bedrock_invoke or is_vertex_ai + _anthropic_content_element = create_anthropic_image_param( + image_url_input, + format=format, + is_bedrock_invoke=force_base64, + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_content_element, + original_content_element=dict(m), + ) + + if "cache_control" in _content_element: + _anthropic_content_element["cache_control"] = _content_element["cache_control"] + user_content.append(_anthropic_content_element) + elif m.get("type", "") == "text": + m = cast(ChatCompletionTextObject, m) + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=m["text"], + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(m), + ) + _content_element = cast(AnthropicMessagesTextParam, _content_element) + + user_content.append(_content_element) + elif m.get("type", "") == "document": + _document_content_element = cast( + AnthropicMessagesDocumentParam, + add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, m), + original_content_element=dict(m), + ), + ) + user_content.append(_document_content_element) + elif m.get("type", "") == "file": + _file_content_element = anthropic_process_openai_file_message( + cast(ChatCompletionFileObject, m) + ) + _file_content_element = add_cache_control_to_content( + anthropic_content_element=cast( + AnthropicMessagesDocumentParam, + _file_content_element, + ), + original_content_element=dict(m), + ) + user_content.append( + cast( + AnthropicMessagesDocumentParam, + _file_content_element, + ) + ) + if ( + len(user_content) > user_content_start + and user_message_types_block.get("cache_control") is not None + and not any( + "cache_control" in content and content["cache_control"] is not None + for content in user_content[user_content_start:] + ) + ): + cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], cached_user_content + ), + original_content_element=user_message_types_block, + ) + user_content[-1] = cached_user_content + elif isinstance(user_message_types_block["content"], str): + _anthropic_content_text_element: AnthropicMessagesTextParam = { + "type": "text", + "text": user_message_types_block["content"], + } + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_content_text_element, + original_content_element=dict(user_message_types_block), + ) + + if "cache_control" in _content_element: + _anthropic_content_text_element["cache_control"] = _content_element["cache_control"] + + user_content.append(_anthropic_content_text_element) + + elif user_message_types_block["role"] == "tool" or user_message_types_block["role"] == "function": + # OpenAI's tool message content will always be a string + user_content.append( + convert_to_anthropic_tool_result(user_message_types_block, force_base64=force_base64) + ) + + msg_i += 1 + + if user_content: + new_messages.append({"role": "user", "content": user_content}) + + # Track unique tool IDs in this merge block to avoid duplication + unique_tool_ids: set[str] = set() + + assistant_content: list[AnthropicMessagesAssistantMessageValues] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_content_block: ChatCompletionAssistantMessage = messages[msg_i] + + # Extract compaction_blocks from provider_specific_fields and add them first + _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") + if isinstance(_provider_specific_fields_raw, dict): + _compaction_blocks = _provider_specific_fields_raw.get("compaction_blocks") + if _compaction_blocks and isinstance(_compaction_blocks, list): + # Add compaction blocks at the beginning of assistant content : https://platform.claude.com/docs/en/build-with-claude/compaction + assistant_content.extend(_compaction_blocks) + + _raw_thinking_blocks = assistant_content_block.get("thinking_blocks", None) + thinking_blocks = ( + _drop_unsignable_thinking_blocks(_raw_thinking_blocks) if _raw_thinking_blocks is not None else None + ) + + # Check if tool_calls contain server tool calls (web search, etc.) + # If so, we need to interleave thinking blocks with tool call groups + # to preserve the original content block ordering. + # Fixes: https://github.com/BerriAI/litellm/issues/23047 + assistant_tool_calls = assistant_content_block.get("tool_calls") + _has_server_tool_calls = False + if assistant_tool_calls is not None: + for _tc in assistant_tool_calls: + _tc_id = _tc.get("id") if isinstance(_tc, dict) else getattr(_tc, "id", None) + if _tc_id and isinstance(_tc_id, str) and _tc_id.startswith("srvtoolu_"): + _has_server_tool_calls = True + break + + _content_is_list: bool = "content" in assistant_content_block and isinstance( + assistant_content_block["content"], list + ) + _content_list = assistant_content_block.get("content") if _content_is_list else None + assistant_content_start: int = len(assistant_content) + + if ( + thinking_blocks is not None + and _has_server_tool_calls + and isinstance(assistant_content_block.get("content", None), (str, type(None))) + ): + # INTERLEAVED MODE: When we have both thinking blocks and server + # tool calls (e.g. web search), Anthropic's original response + # interleaves them: [thinking_1, server_tool_use_1, result_1, + # thinking_2, text, server_tool_use_2, result_2, ...]. + # We must preserve this interleaved order because Anthropic + # verifies thinking block signatures based on position. + + # Build the tool call groups (server_tool_use + its result) + _provider_specific_fields_raw_tc = assistant_content_block.get("provider_specific_fields") + _provider_specific_fields_tc: dict[str, Any] = {} + if isinstance(_provider_specific_fields_raw_tc, dict): + _provider_specific_fields_tc = cast(dict[str, Any], _provider_specific_fields_raw_tc) + _web_search_results_tc = _provider_specific_fields_tc.get("web_search_results") + _tool_results_tc = _provider_specific_fields_tc.get("tool_results") + tool_invoke_results = convert_to_anthropic_tool_invoke( + assistant_tool_calls, + web_search_results=_web_search_results_tc, + tool_results=_tool_results_tc, + ) + + # Group tool invoke results into (server_tool_use, result) pairs + # and separate regular tool_use blocks + server_tool_groups: list[list[Any]] = [] + regular_tool_uses: list[Any] = [] + _current_group: list[Any] = [] + for item in tool_invoke_results: + item_type = item.get("type", "") if isinstance(item, dict) else getattr(item, "type", "") + if item_type == "server_tool_use": + if _current_group: + server_tool_groups.append(_current_group) + _current_group = [item] + elif item_type.endswith("_tool_result"): + _current_group.append(item) + elif item_type == "tool_use": + regular_tool_uses.append(item) + else: + _current_group.append(item) + if _current_group: + server_tool_groups.append(_current_group) + + # Build the text block if content is a non-empty string + text_element = None + _acb_content = assistant_content_block.get("content") + if isinstance(_acb_content, str) and _acb_content: + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=_acb_content, + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(assistant_content_block), + ) + if "cache_control" in _content_element: + _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] + text_element = _anthropic_text_content_element + + # Interleave: each thinking block precedes its server tool group. + # Pattern: thinking[0], group[0], thinking[1], group[1], ... + # Any remaining thinking blocks (after all groups) go before text. + # Any remaining groups (after all thinking blocks) go after. + tb_idx = 0 + grp_idx = 0 + num_tb = len(thinking_blocks) if thinking_blocks else 0 + num_grp = len(server_tool_groups) + + while tb_idx < num_tb or grp_idx < num_grp: + if tb_idx < num_tb and grp_idx < num_grp: + # Emit thinking block then its tool group + assistant_content.append(thinking_blocks[tb_idx]) + tb_idx += 1 + for block in server_tool_groups[grp_idx]: + item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) + grp_idx += 1 + elif tb_idx < num_tb: + # More thinking blocks than tool groups - emit before text + assistant_content.append(thinking_blocks[tb_idx]) + tb_idx += 1 + else: + # More tool groups than thinking blocks - emit remaining + for block in server_tool_groups[grp_idx]: + item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) + grp_idx += 1 + + # Add text block (if any) + if text_element is not None: + assistant_content.append(text_element) + + # Add regular (non-server) tool calls at the end + for item in regular_tool_uses: + item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) + + # Mark tool_calls as already processed so they are not added again + assistant_tool_calls = None + + else: + # SEQUENTIAL MODE: No server tool calls, or no thinking blocks, + # or content is a list. Use the original sequential approach. + + # When content is a list, check if it already contains thinking + # blocks inline. If so, skip prepending thinking_blocks to avoid + # duplication and preserve the original interleaved order. + # Fixes the gap where list-content messages bypass INTERLEAVED + # MODE and still get thinking blocks prepended out of order. + _list_has_thinking = False + if _content_is_list and _content_list is not None: + for _item in _content_list: + if isinstance(_item, dict) and _item.get("type") in ( + "thinking", + "redacted_thinking", + ): + _list_has_thinking = True + break + + if ( + thinking_blocks is not None and not _list_has_thinking + ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR + assistant_content.extend(thinking_blocks) + if _content_is_list and _content_list is not None: + assistant_content_start = len(assistant_content) + for m in _content_list: + if not isinstance(m, dict): + continue + # handle thinking blocks + thinking_block = cast(str, m.get("thinking", "")) + text_block = cast(str, m.get("text", "")) + if ( + m.get("type", "") == "thinking" + and len(thinking_block) > 0 + and not is_unsignable_thinking_block(m) + ): # don't pass empty text blocks. anthropic api raises errors. + anthropic_message: ChatCompletionThinkingBlock | AnthropicMessagesTextParam = cast( + ChatCompletionThinkingBlock, m + ) + assistant_content.append(anthropic_message) + # handle text + elif ( + m.get("type", "") == "text" and len(text_block) > 0 + ): # don't pass empty text blocks. anthropic api raises errors. + anthropic_message = AnthropicMessagesTextParam(type="text", text=text_block) + _cached_message = add_cache_control_to_content( + anthropic_content_element=anthropic_message, + original_content_element=dict(m), + ) + + assistant_content.append(cast(AnthropicMessagesTextParam, _cached_message)) + # handle server_tool_use blocks (tool search, web search, etc.) + # Pass through as-is since these are Anthropic-native content types + elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): + assistant_content.append(m) + elif ( + "content" in assistant_content_block + and isinstance(assistant_content_block["content"], str) + and assistant_content_block["content"] # don't pass empty text blocks. anthropic api raises errors. + ): + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=assistant_content_block["content"], + ) + + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(assistant_content_block), + ) + + if "cache_control" in _content_element: + _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] + + assistant_content.append(_anthropic_text_content_element) + + if assistant_tool_calls is not None: # support assistant tool invoke conversion + # Get web_search_results and tool_results from provider_specific_fields + # for server_tool_use reconstruction. + # Fixes: https://github.com/BerriAI/litellm/issues/17737 + _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") + _provider_specific_fields: dict[str, Any] = {} + if isinstance(_provider_specific_fields_raw, dict): + _provider_specific_fields = cast(dict[str, Any], _provider_specific_fields_raw) + _web_search_results = _provider_specific_fields.get("web_search_results") + _tool_results = _provider_specific_fields.get("tool_results") + tool_invoke_results = convert_to_anthropic_tool_invoke( + assistant_tool_calls, + web_search_results=_web_search_results, + tool_results=_tool_results, + ) + + # Prevent "tool_use ids must be unique" errors by filtering duplicates + # This can happen when merging history that already contains the tool calls + for item in tool_invoke_results: + # tool_use items are typically dicts, but handle objects just in case + item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) + + if item_id: + if item_id in unique_tool_ids: + continue + unique_tool_ids.add(item_id) + + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) + + assistant_function_call = assistant_content_block.get("function_call") + + if assistant_function_call is not None: + assistant_content.extend(convert_function_to_anthropic_tool_invoke(assistant_function_call)) + + if ( + _content_is_list + and len(assistant_content) > assistant_content_start + and assistant_content_block.get("cache_control") is not None + and not any( + "cache_control" in content and content["cache_control"] is not None + for content in assistant_content[assistant_content_start:] + ) + ): + cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], cached_assistant_content + ), + original_content_element=assistant_content_block, + ) + assistant_content[-1] = cached_assistant_content + + msg_i += 1 + + if assistant_content: + new_messages.append({"role": "assistant", "content": assistant_content}) + + ## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + new_messages.extend(anthropic_system_messages(messages[msg_i])) + msg_i += 1 + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + if len(new_messages) > 0 and new_messages[-1]["role"] == "assistant": + if isinstance(new_messages[-1]["content"], str): + new_messages[-1]["content"] = new_messages[-1]["content"].rstrip() + elif isinstance(new_messages[-1]["content"], list): + for content in new_messages[-1]["content"]: + if isinstance(content, dict) and content["type"] == "text": + content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message + + return new_messages + + +def extract_between_tags(tag: str, string: str, strip: bool = False) -> list[str]: + ext_list = re.findall(f"<{tag}>(.+?)", string, re.DOTALL) + if strip: + ext_list = [e.strip() for e in ext_list] + return ext_list + + +def contains_tag(tag: str, string: str) -> bool: + return bool(re.search(f"<{tag}>(.+?)", string, re.DOTALL)) + + +def parse_xml_params(xml_content, json_schema: dict | None = None): + """ + Compare the xml output to the json schema + + check if a value is a list - if so, get it's child elements + """ + root: Final = ET.fromstring(xml_content) + params: Final = {} + + if json_schema is not None: # check if we have a json schema for this function call + # iterate over all properties in the schema + for prop in json_schema["properties"]: + # If property is an array, get the nested items + _element = root.find(f"parameters/{prop}") + if json_schema["properties"][prop]["type"] == "array": + items = [] + if _element is not None: + for value in _element: + try: + if value.text is not None: + _value = json.loads(value.text) + else: + continue + except json.JSONDecodeError: + _value = value.text + items.append(_value) + params[prop] = items + # If property is not an array, append the value directly + elif _element is not None and _element.text is not None: + try: + _value = json.loads(_element.text) + except json.JSONDecodeError: + _value = _element.text + params[prop] = _value + else: + for child in root.findall(".//parameters/*"): + if child is not None and child.text is not None: + try: + # Attempt to decode the element's text as JSON + params[child.tag] = json.loads(child.text) + except json.JSONDecodeError: + # If JSON decoding fails, use the original text + params[child.tag] = child.text + + return params + + +### GEMINI HELPER FUNCTIONS ### + + +def get_system_prompt(messages): + system_prompt_indices: Final = [] + system_prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "system": + system_prompt += message["content"] + system_prompt_indices.append(idx) + if len(system_prompt_indices) > 0: + for idx in reversed(system_prompt_indices): + messages.pop(idx) + return system_prompt, messages + + +from litellm.types.llms.cohere import ( + CallObject, + ChatHistory, + ChatHistoryChatBot, + ChatHistorySystem, + ChatHistoryToolResult, + ChatHistoryUser, + ToolCallObject, + ToolResultObject, +) + + +def convert_openai_message_to_cohere_tool_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + tool_calls: list, +) -> ToolResultObject: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "content": {"location": "San Francisco, CA", "unit": "fahrenheit", "temperature": "72"}, + }, + """ + """ + OpenAI message with a function call looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + + """ + Cohere tool_results look like: + { + "call": { + "name": "query_daily_sales_report", + "parameters": { + "day": "2023-09-29" + }, + }, + "outputs": [ + { + "date": "2023-09-29", + "summary": "Total Sales Amount: 10000, Total Units Sold: 250" + } + ] + }, + """ + + content_str: str = "" + if isinstance(message["content"], str): + content_str = message["content"] + elif isinstance(message["content"], list): + content_list: Final = message["content"] + for content in content_list: + if content["type"] == "text": + content_str += content["text"] + if len(content_str) > 0: + try: + content = json.loads(content_str) + except json.JSONDecodeError: + content = {"result": content_str} + else: + content = {} + name = "" + arguments = {} + # Recover name from last message with tool calls + if len(tool_calls) > 0: + tools: Final = tool_calls + msg_tool_call_id: Final = message.get("tool_call_id", None) + for tool in tools: + prev_tool_call_id = tool.get("id", None) + if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: + name = tool.get("function", {}).get("name", "") + arguments_str = tool.get("function", {}).get("arguments", "") + if arguments_str is not None and len(arguments_str) > 0: + arguments = json.loads(arguments_str) + + if message["role"] == "function": + function_message: Final[ChatCompletionFunctionMessage] = message + name = function_message["name"] + cohere_tool_result: ToolResultObject = { + "call": CallObject(name=name, parameters=arguments), + "outputs": [content], + } + return cohere_tool_result + else: + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + + cohere_tool_result = { + "call": CallObject(name=name, parameters=arguments), + "outputs": [content], + } + return cohere_tool_result + + +def get_all_tool_calls(messages: list) -> list: + """ + Returns extracted list of `tool_calls`. + + Done to handle openai no longer returning tool call 'name' in tool results. + """ + tool_calls: Final[list] = [] + for m in messages: + if m.get("tool_calls", None) is not None: + if isinstance(m["tool_calls"], list): + tool_calls.extend(m["tool_calls"]) + + return tool_calls + + +def convert_to_cohere_tool_invoke(tool_calls: list) -> list[ToolCallObject]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + + """ + Cohere tool invokes: + { + "role": "CHATBOT", + "tool_calls": [{"name": "get_weather", "parameters": {"location": "San Francisco, CA"}}] + } + """ + + cohere_tool_invoke: Final[list[ToolCallObject]] = [ + { + "name": get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), + "parameters": json.loads(get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments")), + } + for tool in tool_calls + if get_attribute_or_key(tool, "type") == "function" + ] + + return cohere_tool_invoke + + +def cohere_messages_pt_v2( + messages: list, + model: str, + llm_provider: str, +) -> tuple[str | ToolResultObject, ChatHistory]: + """ + Returns a tuple(Union[tool_result, message], chat_history) + + - if last message is tool result -> return 'tool_result' + - if last message is text -> return message (str) + + - return preceding messages as 'chat_history' + + Note: + - cannot specify message if the last entry in chat history contains tool results + - message must be at least 1 token long or tool results must be specified. + - cannot specify tool_results if the last entry in chat history contains a user message + """ + tool_calls: Final[list] = get_all_tool_calls(messages=messages) + + ## GET MOST RECENT MESSAGE + most_recent_message: Final = messages.pop(-1) + returned_message: ToolResultObject | str = "" + if most_recent_message.get("role", "") is not None and most_recent_message["role"] == "tool": + # tool result + returned_message = convert_openai_message_to_cohere_tool_result(most_recent_message, tool_calls) + else: + content: Final[str | list] = most_recent_message.get("content") + if isinstance(content, str): + returned_message = content + else: + for chunk in content: + if chunk.get("type") == "text": + returned_message += chunk.get("text") + + ## CREATE CHAT HISTORY + user_message_types: Final = {"user"} + tool_message_types: Final = {"tool", "function"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final[ChatHistory] = [] + msg_i = 0 + + while msg_i < len(messages): + user_content: str = "" + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + user_content += m["text"] + else: + user_content += messages[msg_i]["content"] + msg_i += 1 + + if len(user_content) > 0: + new_messages.append(ChatHistoryUser(role="USER", message=user_content)) + + system_content: str = "" + ## MERGE CONSECUTIVE SYSTEM CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + system_content += m["text"] + else: + system_content += messages[msg_i]["content"] + msg_i += 1 + + if len(system_content) > 0: + new_messages.append(ChatHistorySystem(role="SYSTEM", message=system_content)) + + assistant_content: str = "" + assistant_tool_calls: list[ToolCallObject] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + if messages[msg_i].get("content", None) is not None and isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + assistant_content += m["text"] + elif messages[msg_i].get("content") is not None and isinstance(messages[msg_i]["content"], str): + assistant_content += messages[msg_i]["content"] + if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion + assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["tool_calls"])) + + if messages[msg_i].get("function_call"): + assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["function_call"])) + + msg_i += 1 + + if len(assistant_content) > 0: + new_messages.append( + ChatHistoryChatBot( + role="CHATBOT", + message=assistant_content, + tool_calls=assistant_tool_calls, + ) + ) + + ## MERGE CONSECUTIVE TOOL RESULTS + tool_results: list[ToolResultObject] = [] + while msg_i < len(messages) and messages[msg_i]["role"] in tool_message_types: + tool_results.append(convert_openai_message_to_cohere_tool_result(messages[msg_i], tool_calls)) + + msg_i += 1 + + if len(tool_results) > 0: + new_messages.append(ChatHistoryToolResult(role="TOOL", tool_results=tool_results)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return returned_message, new_messages + + +def cohere_message_pt(messages: list): + tool_calls: Final[list] = get_all_tool_calls(messages=messages) + prompt = "" + tool_results: Final = [] + for message in messages: + # check if this is a tool_call result + if message["role"] == "tool": + tool_result = convert_openai_message_to_cohere_tool_result(message, tool_calls=tool_calls) + tool_results.append(tool_result) + elif message.get("content"): + prompt += message["content"] + "\n\n" + prompt = prompt.rstrip() + return prompt, tool_results + + +def amazon_titan_pt( + messages: list, +): # format - https://github.com/BerriAI/litellm/issues/1896 + """ + Amazon Titan uses 'User:' and 'Bot: in it's prompt template + """ + + class AmazonTitanConstants(Enum): + HUMAN_PROMPT = "\n\nUser: " # Assuming this is similar to Anthropic prompt formatting, since amazon titan's prompt formatting is currently undocumented + AI_PROMPT = "\n\nBot: " + + prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "user": + prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" + else: + prompt += f"{AmazonTitanConstants.AI_PROMPT.value}{message['content']}" + if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` + prompt = f"{AmazonTitanConstants.HUMAN_PROMPT.value}" + prompt + if messages[-1]["role"] != "assistant": + prompt += f"{AmazonTitanConstants.AI_PROMPT.value}" + return prompt + + +def _load_image_from_url(image_url): + try: + from PIL import Image + except Exception: + raise Exception("image conversion failed please run `pip install Pillow`") + from io import BytesIO + + try: + # Send a GET request to the image URL + client: Final = HTTPHandler(concurrent_limit=1) + response: Final[httpx.Response] = safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + # Check the response's content type to ensure it is an image + content_type: Final = response.headers.get("content-type") + if not content_type or "image" not in content_type: + raise ValueError(f"URL does not point to a valid image (content-type: {content_type})") + + # Load the image from the response content + return Image.open(BytesIO(response.content)) + + except Exception as e: + raise e + + +def _gemini_vision_convert_messages(messages: list): + """ + Converts given messages for GPT-4 Vision to Gemini format. + + Args: + messages (list): The messages to convert. Each message can be a dictionary with a "content" key. The content can be a string or a list of elements. If it is a string, it will be concatenated to the prompt. If it is a list, each element will be processed based on its type: + - If the element is a dictionary with a "type" key equal to "text", its "text" value will be concatenated to the prompt. + - If the element is a dictionary with a "type" key equal to "image_url", its "image_url" value will be added to the list of images. + + Returns: + tuple: A tuple containing the prompt (a string) and the processed images (a list of objects representing the images). + """ + + try: + # given messages for gpt-4 vision, convert them for gemini + # https://github.com/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_python.ipynb + prompt = "" + images: Final = [] + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + elif element["type"] == "image_url": + image_url = element["image_url"]["url"] + images.append(image_url) + # processing images passed to gemini + processed_images: Final = [] + for img in images: + if "https:/" in img: + # Case 1: Image from URL + image = _load_image_from_url(img) + processed_images.append(image) + + else: + try: + from PIL import Image + except Exception: + raise Exception("gemini image conversion failed please run `pip install Pillow`") + + if "base64" in img: + # Case 2: Base64 image data + import base64 + import io + + # Extract the base64 image data + base64_data = img.split("base64,")[1] + + # Decode the base64 image data + image_data = base64.b64decode(base64_data) + + # Load the image from the decoded data + image = Image.open(io.BytesIO(image_data)) + else: + # Case 3: Image filepath (e.g. temp.jpeg) given + image = Image.open(img) + processed_images.append(image) + content: Final = [prompt] + processed_images + return content + except Exception as e: + raise e + + +def gemini_text_image_pt(messages: list): + """ + { + "contents":[ + { + "parts":[ + {"text": "What is this picture?"}, + { + "inline_data": { + "mime_type":"image/jpeg", + "data": "'$(base64 -w0 image.jpg)'" + } + } + ] + } + ] + } + """ + try: + pass + except Exception: + raise Exception("Importing google.generativeai failed, please run 'pip install -q google-generativeai") + + prompt = "" + images: Final = [] + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + elif element["type"] == "image_url": + image_url = element["image_url"]["url"] + images.append(image_url) + + content: Final = [prompt] + images + return content + + +def azure_text_pt(messages: list): + prompt = "" + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + return prompt + + +###### AZURE AI ####### +def stringify_json_tool_call_content(messages: list) -> list: + """ + + - Check 'content' in tool role -> convert to dict (if not) -> stringify + + Done for azure_ai/cohere calls to handle results of a tool call + """ + + for m in messages: + if m["role"] == "tool" and isinstance(m["content"], str): + # check if content is a valid json object + try: + json.loads(m["content"]) + except json.JSONDecodeError: + m["content"] = json.dumps({"result": m["content"]}) + + return messages + + +###### AMAZON BEDROCK ####### + +from email.message import Message + +import httpx + +from litellm.types.llms.bedrock import ( + BedrockConverseReasoningContentBlock, + BedrockConverseReasoningTextBlock, + BedrockToolSpec, + SearchResultBlock, +) +from litellm.types.llms.bedrock import ContentBlock as BedrockContentBlock +from litellm.types.llms.bedrock import DocumentBlock as BedrockDocumentBlock +from litellm.types.llms.bedrock import ImageBlock as BedrockImageBlock +from litellm.types.llms.bedrock import SourceBlock as BedrockSourceBlock +from litellm.types.llms.bedrock import ToolBlock as BedrockToolBlock +from litellm.types.llms.bedrock import ToolResultBlock as BedrockToolResultBlock +from litellm.types.llms.bedrock import ( + ToolResultContentBlock as BedrockToolResultContentBlock, +) +from litellm.types.llms.bedrock import ToolUseBlock as BedrockToolUseBlock +from litellm.types.llms.bedrock import VideoBlock as BedrockVideoBlock + + +def _parse_content_type(content_type: str) -> str: + m: Final = Message() + m["content-type"] = content_type + return m.get_content_type() + + +def _parse_mime_type(base64_data: str) -> str | None: + mime_type_match: Final = re.match(r"data:(.*?);base64", base64_data) + if mime_type_match: + return mime_type_match.group(1) + else: + return None + + +class BedrockImageProcessor: + """Handles both sync and async image processing for Bedrock conversations.""" + + @staticmethod + def _post_call_image_processing(response: httpx.Response, image_url: str = "") -> tuple[str, str]: + # Check the response's content type to ensure it is an image + content_type: str | None = response.headers.get("content-type") + + # Use helper function to infer content type with fallback logic + content_type = infer_content_type_from_url_and_content( + url=image_url, + content=response.content, + current_content_type=content_type, + ) + + content_type = _parse_content_type(content_type) + + # Convert the image content to base64 bytes + base64_bytes: Final = base64.b64encode(response.content).decode("utf-8") + + return base64_bytes, content_type + + @staticmethod + async def get_image_details_async(image_url) -> tuple[str, str]: + try: + client: Final = get_async_httpx_client( + llm_provider=httpxSpecialProvider.PromptFactory, + params={"concurrent_limit": 1}, + ) + # Send a GET request to the image URL + response: Final[httpx.Response] = await async_safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + return BedrockImageProcessor._post_call_image_processing(response, image_url) + + except Exception as e: + raise e + + @staticmethod + def get_image_details(image_url) -> tuple[str, str]: + try: + client: Final = HTTPHandler(concurrent_limit=1) + # Send a GET request to the image URL + response: Final[httpx.Response] = safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + return BedrockImageProcessor._post_call_image_processing(response, image_url) + + except Exception as e: + raise e + + @staticmethod + def _parse_base64_image(image_url: str) -> tuple[str, str, str]: + """Parse base64 encoded image data.""" + image_metadata, img_without_base_64 = image_url.split(",") + + # Extract MIME type using regular expression + mime_type_match: Final = re.match(r"data:(.*?);base64", image_metadata) + + if mime_type_match: + mime_type = mime_type_match.group(1) + mime_type = mime_type.split(";")[0] + image_format = mime_type.split("/")[1] + else: + mime_type = "image/jpeg" + image_format = "jpeg" + + return img_without_base_64, mime_type, image_format + + @staticmethod + def _validate_format(mime_type: str, image_format: str) -> str: + """Validate image format and mime type for both images and documents.""" + + supported_image_formats: Final = litellm.AmazonConverseConfig().get_supported_image_types() + supported_doc_formats: Final = litellm.AmazonConverseConfig().get_supported_document_types() + supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() + + document_types: Final = ["application", "text"] + is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) + + supported_image_and_video_formats: Final[list[str]] = supported_video_formats + supported_image_formats + + if is_document: + return BedrockImageProcessor._get_document_format( + mime_type=mime_type, supported_doc_formats=supported_doc_formats + ) + + else: + ######################################################### + # Check if image_format is an image or video + ######################################################### + if image_format not in supported_image_and_video_formats: + raise ValueError( + f"Unsupported image format: {image_format}. Supported formats: {supported_image_and_video_formats}" + ) + return image_format + + @staticmethod + def _get_document_format(mime_type: str, supported_doc_formats: list[str]) -> str: + """ + Get the document format from the mime type + + - Primary method - uses `mimetypes.guess_all_extensions` + - Fallback method - uses `get_file_extension_from_mime_type` + + Relevant Issue: https://github.com/BerriAI/litellm/issues/12260 + + `mimetypes` is not available in docker containers, so we fallback to `get_file_extension_from_mime_type` + + Args: + mime_type: The mime type of the document + supported_doc_formats: The supported document formats for the current model + + Returns: + The document format + """ + valid_extensions: list[str] | None = None + potential_extensions: Final = mimetypes.guess_all_extensions(mime_type, strict=False) + valid_extensions = [ext[1:] for ext in potential_extensions if ext[1:] in supported_doc_formats] + + # Fallback to types/files.py if mimetypes doesn't return valid extensions + ################# + # litellm runs on docker containers and `mimetypes` depends on the installed mimetypes of the OS + # we fallback to well known mime types in types/files.py if mimetypes doesn't return valid extensions + if not valid_extensions: + try: + fallback_extension: Final = get_file_extension_from_mime_type(mime_type) + if fallback_extension in supported_doc_formats: + valid_extensions = [fallback_extension] + except ValueError: + # Neither mimetypes nor files.py could handle this MIME type + # get_file_extension_from_mime_type raises ValueError if the mime type is not supported + pass + + if not valid_extensions: + raise ValueError( + f"No supported extensions for MIME type: {mime_type}. Supported formats: {supported_doc_formats}" + ) + + # Use first valid extension instead of provided image_format + return valid_extensions[0] + + @staticmethod + def _create_bedrock_block(image_bytes: str, mime_type: str, image_format: str) -> BedrockContentBlock: + """Create appropriate Bedrock content block based on mime type.""" + _blob: Final = BedrockSourceBlock(bytes=image_bytes) + + document_types: Final = ["application", "text"] + is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) + + supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() + is_video: Final = any(image_format.startswith(video_type) for video_type in supported_video_formats) + + HASH_SAMPLE_BYTES: Final = 64 * 1024 # hash up to 64 KB of data + + if is_document: + # --- Prepare normalized bytes for hashing (without modifying original) --- + if isinstance(image_bytes, str): + # Remove whitespace/newlines so base64 variations hash identically + normalized = "".join(image_bytes.split()).encode("utf-8") + else: + normalized = image_bytes + + # --- Use only the first 64 KB for speed --- + if len(normalized) <= HASH_SAMPLE_BYTES: + sample = normalized + else: + sample = normalized[:HASH_SAMPLE_BYTES] + + # --- Compute deterministic hash (sample + total length) --- + hasher: Final = hashlib.sha256() + hasher.update(sample) + hasher.update(str(len(normalized)).encode("utf-8")) # include full length for uniqueness + full_hash: Final = hasher.hexdigest() + content_hash: Final = full_hash[:16] # short deterministic ID + + document_name: Final = f"DocumentPDFmessages_{content_hash}_{image_format}" + + # --- Return content block --- + return BedrockContentBlock( + document=BedrockDocumentBlock( + source=_blob, + format=image_format, + name=document_name, + ) + ) + elif is_video: + return BedrockContentBlock(video=BedrockVideoBlock(source=_blob, format=image_format)) + else: + return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) + + @classmethod + def process_image_sync(cls, image_url: str, format: str | None = None) -> BedrockContentBlock: + """Synchronous image processing.""" + + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "http://" in image_url or "https://" in image_url: + img_bytes, mime_type = BedrockImageProcessor.get_image_details(image_url) + image_format = mime_type.split("/")[1] + else: + raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") + + if format: + mime_type = format + image_format = mime_type.split("/")[1] + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) + + @classmethod + async def process_image_async(cls, image_url: str, format: str | None) -> BedrockContentBlock: + """Asynchronous image processing.""" + + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "http://" in image_url or "https://" in image_url: + img_bytes, mime_type = await BedrockImageProcessor.get_image_details_async(image_url) + image_format = mime_type.split("/")[1] + else: + raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") + + if format: # override with user-defined params + mime_type = format + image_format = mime_type.split("/")[1] + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) + + +def _convert_to_bedrock_tool_call_invoke( + tool_calls: list, + model: str | None = None, +) -> list[BedrockContentBlock]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + """ + Bedrock tool invokes: + [ + { + "role": "assistant", + "toolUse": { + "input": {"location": "Boston, MA", ..}, + "name": "get_current_weather", + "toolUseId": "call_abc123" + } + } + ] + """ + """ + - json.loads argument + - extract name + - extract id + """ + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + split_concatenated_json_objects, + ) + + try: + _parts_list: Final[list[BedrockContentBlock]] = [] + for tool in tool_calls: + if "function" in tool: + tool_id = tool["id"] + name = make_valid_bedrock_tool_name(tool["function"].get("name", "")) + arguments = tool["function"].get("arguments", "") + + if not arguments or not arguments.strip(): + arguments_dict = {} + else: + try: + arguments_dict = json.loads(arguments) + # Ensure arguments_dict is always a dict + # (Bedrock requires toolUse.input to be an object). + # Some providers return arguments: '""' which + # json.loads decodes to a bare string. + if not isinstance(arguments_dict, dict): + arguments_dict = {} + except json.JSONDecodeError: + # The model may return multiple JSON objects + # concatenated in a single arguments string, e.g. + # '{"cmd":"a"}{"cmd":"b"}{"cmd":"c"}' + # Split them and emit one toolUse block per object. + # Fixes: https://github.com/BerriAI/litellm/issues/20543 + parsed_objects = split_concatenated_json_objects(arguments) + if parsed_objects: + # First object keeps the original tool id. + for obj_idx, obj in enumerate(parsed_objects): + block_id = _sanitize_bedrock_tool_use_id( + tool_id if obj_idx == 0 else f"{tool_id}_{obj_idx}" + ) + bedrock_tool = BedrockToolUseBlock(input=obj, name=name, toolUseId=block_id) + _parts_list.append(BedrockContentBlock(toolUse=bedrock_tool)) + # cache_control applies to the whole original + # tool call; attach after the last split block. + if tool.get("cache_control", None) is not None: + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool["cache_control"]}, + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts_list.append(_cache_point_block) + continue + # Fallback: no objects extracted — use empty dict. + arguments_dict = {} + + bedrock_tool = BedrockToolUseBlock( + input=arguments_dict, name=name, toolUseId=_sanitize_bedrock_tool_use_id(tool_id) + ) + bedrock_content_block = BedrockContentBlock(toolUse=bedrock_tool) + _parts_list.append(bedrock_content_block) + + # Check for cache_control and add a separate cachePoint block + if tool.get("cache_control", None) is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool["cache_control"]}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + _parts_list.append(cache_point_block) + return _parts_list + except Exception as e: + tool_call_ids: Final = tuple(tool.get("id") for tool in tool_calls if isinstance(tool, dict)) + raise litellm.BadRequestError( + message=f"Unable to convert openai tool calls with ids={tool_call_ids} to bedrock tool calls. " + f"Received error={e}", + model=model or "", + llm_provider="bedrock", + ) from e + + +def _append_bedrock_tool_result_media_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + processed_block: BedrockContentBlock, + content: dict, + content_type: str, +) -> None: + if "image" in processed_block: + tool_result_content_blocks.append(BedrockToolResultContentBlock(image=processed_block["image"])) + elif "document" in processed_block: + tool_result_content_blocks.append(BedrockToolResultContentBlock(document=processed_block["document"])) + else: + verbose_logger.warning( + "Bedrock Converse: unrecognized BedrockContentBlock keys %s for %s tool-result block %s; dropping.", + list(processed_block.keys()), + content_type, + content, + ) + + +def _append_bedrock_tool_result_image_url_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + content: dict, +) -> None: + format: str | None = None + if isinstance(content["image_url"], dict): + image_url = content["image_url"]["url"] + format = content["image_url"].get("format") + else: + image_url = content["image_url"] + processed_block: Final = BedrockImageProcessor.process_image_sync( + image_url=image_url, + format=format, + ) + _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "image_url") + + +def _append_bedrock_tool_result_file_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + content: dict, +) -> None: + # Match the user-message path (_process_file_message): accept either + # file_data (base64 data URI) or file_id (server-side reference / URL). + file_obj: Final = content.get("file") or {} + file_data: Final = file_obj.get("file_data") + file_id: Final = file_obj.get("file_id") + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={content}", + model="", + llm_provider="bedrock", + ) + processed_block: Final = BedrockImageProcessor.process_image_sync( + image_url=cast(str, file_id or file_data), + format=file_obj.get("format"), + ) + _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "file") + + +def _parse_bedrock_tool_result_content_list( + content_list: list, +) -> list[BedrockToolResultContentBlock]: + tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] + for content in content_list: + if content["type"] == "text": + tool_result_content_blocks.append(BedrockToolResultContentBlock(text=content["text"])) + elif content["type"] == "image_url": + _append_bedrock_tool_result_image_url_block(tool_result_content_blocks, content) + elif content["type"] == "file": + _append_bedrock_tool_result_file_block(tool_result_content_blocks, content) + return tool_result_content_blocks + + +def _build_bedrock_tool_result_content_blocks( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, +) -> tuple[list[BedrockToolResultContentBlock], bool]: + # Optional OpenAI tool-message extension: + # allow structured Bedrock search results on tool messages and map them + # directly to toolResult.content[].searchResult for Converse API. + # + # If `search_results` is present, we intentionally prefer it over `content` + # to avoid generating mixed text + searchResult blocks. + search_results: Final = message.get("search_results") + if isinstance(search_results, list): + tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] + for result in search_results: + if not isinstance(result, dict): + continue + tool_result_content_blocks.append( + BedrockToolResultContentBlock(searchResult=cast(SearchResultBlock, result)) + ) + if tool_result_content_blocks: + return tool_result_content_blocks, True + + message_content: Final = message["content"] + if isinstance(message_content, str): + return [BedrockToolResultContentBlock(text=message_content)], False + if isinstance(message_content, list): + return _parse_bedrock_tool_result_content_list(message_content), False + return [], False + + +def _convert_to_bedrock_tool_call_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, +) -> BedrockContentBlock: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + """ + Bedrock result looks like this: + { + "role": "user", + "content": [ + { + "toolResult": { + "toolUseId": "tooluse_kZJMlvQmRJ6eAyJE5GIl7Q", + "content": [ + { + "json": { + "song": "Elemental Hotel", + "artist": "8 Storey Hike" + } + } + ] + } + } + ] + } + """ + """ + - + """ + tool_result_content_blocks, used_search_results = _build_bedrock_tool_result_content_blocks(message) + + message.get("name", "") + id: Final = _sanitize_bedrock_tool_use_id(str(message.get("tool_call_id", str(uuid.uuid4())))) + + tool_result: Final = BedrockToolResultBlock(content=tool_result_content_blocks, toolUseId=id) + if used_search_results: + tool_result["status"] = cast(Literal["success"], "success") + + content_block: Final = BedrockContentBlock(toolResult=tool_result) + + return content_block + + +def _deduplicate_bedrock_content_blocks( + blocks: list[BedrockContentBlock], + block_key: str, + id_key: str = "toolUseId", +) -> list[BedrockContentBlock]: + """ + Remove duplicate content blocks that share the same ID under ``block_key``. + + Bedrock requires all toolResult and toolUse IDs within a single message to + be unique. When merging consecutive messages, duplicates can occur if the + same tool_call_id appears multiple times in conversation history. + + When duplicates exist, the first occurrence is retained and subsequent ones + are discarded. A warning is logged for every dropped block so that + upstream duplication bugs remain visible. + + Blocks that do not contain ``block_key`` (e.g., cachePoint, text) are + always preserved. + + Args: + blocks: The list of Bedrock content blocks to deduplicate. + block_key: The dict key to inspect (e.g. ``"toolResult"`` or ``"toolUse"``). + id_key: The nested key that holds the unique ID (default ``"toolUseId"``). + """ + seen_ids: Final[set[str]] = set() + deduplicated: Final[list[BedrockContentBlock]] = [] + for block in blocks: + keyed = block.get(block_key) + if keyed is not None and isinstance(keyed, dict): + block_id = keyed.get(id_key) + if block_id: + if block_id in seen_ids: + verbose_logger.warning( + "Bedrock Converse: dropping duplicate %s block with " + "%s=%s. This may indicate duplicate tool messages in " + "conversation history.", + block_key, + id_key, + block_id, + ) + continue + seen_ids.add(block_id) + deduplicated.append(block) + return deduplicated + + +def _deduplicate_bedrock_tool_content( + tool_content: list[BedrockContentBlock], +) -> list[BedrockContentBlock]: + """Convenience wrapper: deduplicate ``toolResult`` blocks by ``toolUseId``.""" + return _deduplicate_bedrock_content_blocks(tool_content, "toolResult") + + +def _rename_duplicate_bedrock_document_names( + contents: list[BedrockMessageBlock], +) -> list[BedrockMessageBlock]: + """ + Rename duplicate document names across all messages in a Bedrock request. + + Document names are derived from a content hash, so the same file appearing + in multiple conversation turns produces identical names and Bedrock rejects + the request with "Messages can not contain duplicate document names". The + first occurrence keeps its original name so prompt-cache prefixes stay + stable; later occurrences get a deterministic positional suffix + (``_2``, ``_3``, ...), bumped further if the suffixed name already + belongs to another document (e.g. an organic name ending in ``_2``). + """ + used_names: Final[set[str]] = set() + for message in contents: + for block in message.get("content") or []: + document = block.get("document") + if isinstance(document, dict) and document.get("name"): + used_names.add(document["name"]) + + name_counts: Final[dict[str, int]] = {} + for message in contents: + for block in message.get("content") or []: + document = block.get("document") + if not isinstance(document, dict): + continue + name = document.get("name") + if not name: + continue + count = name_counts.get(name, 0) + 1 + name_counts[name] = count + if count > 1: + suffix = count + new_name = f"{name}_{suffix}" + while new_name in used_names: + suffix += 1 + new_name = f"{name}_{suffix}" + used_names.add(new_name) + document["name"] = new_name + return contents + + +BEDROCK_DOCUMENT_PLACEHOLDER_TEXT: Final = "." + + +def _with_text_when_document_only(message: BedrockMessageBlock) -> BedrockMessageBlock: + blocks: Final = message["content"] + needs_text: Final = ( + message["role"] == "user" + and any("document" in block for block in blocks) + and all("text" not in block for block in blocks) + ) + if not needs_text: + return message + placeholder: Final = BedrockContentBlock(text=BEDROCK_DOCUMENT_PLACEHOLDER_TEXT) + cut: Final = len(blocks) - 1 if "cachePoint" in blocks[-1] else len(blocks) + return BedrockMessageBlock(role="user", content=[*blocks[:cut], placeholder, *blocks[cut:]]) + + +def _ensure_document_messages_have_text( + contents: list[BedrockMessageBlock], +) -> list[BedrockMessageBlock]: + """ + Bedrock Converse rejects any user message that carries a document block + without a sibling text block ("A text block must be included when using + documents"), e.g. Claude Code sends the PDF as a document-only user turn. + Inject a placeholder text block, kept ahead of a trailing cachePoint so + the caller's cache boundary stays the final block. + """ + return [_with_text_when_document_only(message) for message in contents] + + +def _sort_bedrock_assistant_content_blocks( + blocks: list[BedrockContentBlock], +) -> list[BedrockContentBlock]: + """ + Sort assistant content blocks so that ``text`` blocks appear before + ``toolUse`` blocks. + + Bedrock requires all ``text`` blocks to precede any ``toolUse`` blocks + within an assistant message. When the Responses API converts + function_call items before message items, the resulting ``toolUse`` + blocks can end up before ``text`` blocks, causing Bedrock to reject + the request with a 400 error because the ``toolUse`` → ``toolResult`` + pairing is broken by the intervening ``text`` block. + + Sort order (stable): + 0 - reasoningContent + 1 - text / image / document / video / other non-tool blocks + 2 - toolUse + """ + + def _sort_key(block: BedrockContentBlock) -> int: + if "reasoningContent" in block: + return 0 + if "toolUse" in block: + return 2 + if "cachePoint" in block: + # cachePoint blocks are paired with their preceding toolUse block. + # Same key as toolUse so Python's stable sort keeps them together. + return 2 + return 1 + + return sorted(blocks, key=_sort_key) + + +def _insert_assistant_continue_message( + messages: list[BedrockMessageBlock], + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> list[BedrockMessageBlock]: + """ + Add dummy message between user/tool result blocks. + + Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 + """ + if assistant_continue_message is not None: + if isinstance(assistant_continue_message, str): + messages.append( + BedrockMessageBlock( + role="assistant", + content=[BedrockContentBlock(text=assistant_continue_message)], + ) + ) + elif isinstance(assistant_continue_message, dict): + text = convert_content_list_to_str(assistant_continue_message) + messages.append( + BedrockMessageBlock( + role="assistant", + content=[BedrockContentBlock(text=text)], + ) + ) + elif litellm.modify_params: + text = convert_content_list_to_str(cast(ChatCompletionAssistantMessage, DEFAULT_ASSISTANT_CONTINUE_MESSAGE)) + messages.append( + BedrockMessageBlock( + role="assistant", + content=[ + BedrockContentBlock(text=text), + ], + ) + ) + return messages + + +def get_user_message_block_or_continue_message( + message: ChatCompletionUserMessage, + user_continue_message: ChatCompletionUserMessage | None = None, +) -> ChatCompletionUserMessage: + """ + Returns the user content block + if content block is an empty string, then return the default continue message + + Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 + """ + content_block: Final = message.get("content", None) + + # Handle None case + if content_block is None or (user_continue_message is None and litellm.modify_params is False): + return skip_empty_text_blocks(message=message) + + # Handle string case + if isinstance(content_block, str): + # check if content is empty + if content_block.strip(): + return message + else: + return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) + + # Handle list case + if isinstance(content_block, list): + """ + CHECK FOR + "content": [ + { + "type": "text", + "text": "" + } + ], + """ + if not content_block: + return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) + # Create a copy of the message to avoid modifying the original + modified_content_block: Final = content_block.copy() + + for item in modified_content_block: + # Check if the list is empty + if item["type"] == "text": + if not item["text"].strip(): + # Replace empty text with continue message + _user_continue_message = ChatCompletionUserMessage( + **(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE) + ) + text = convert_content_list_to_str(_user_continue_message) + item["text"] = text + break + modified_message: Final = message.copy() + modified_message["content"] = modified_content_block + return modified_message + + # Handle unsupported type + raise ValueError(f"Unsupported content type: {type(content_block)}") + + +def return_assistant_continue_message( + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + if assistant_continue_message and isinstance(assistant_continue_message, str): + return ChatCompletionAssistantMessage( + role="assistant", + content=assistant_continue_message, + ) + elif assistant_continue_message and isinstance(assistant_continue_message, dict): + return ChatCompletionAssistantMessage(**assistant_continue_message) + else: + return DEFAULT_ASSISTANT_CONTINUE_MESSAGE + + +def _skip_empty_dict_blocks(blocks: list[dict]) -> list[dict]: + """ + Filter out empty text blocks from a list of dictionaries. + + Args: + blocks: List of dictionaries representing message content blocks + + Returns: + Filtered list of non-empty text blocks + """ + return [item for item in blocks if not (item.get("type") == "text" and not item.get("text", "").strip())] + + +@overload +def skip_empty_text_blocks( + message: ChatCompletionAssistantMessage, +) -> ChatCompletionAssistantMessage: + pass + + +@overload +def skip_empty_text_blocks( + message: ChatCompletionUserMessage, +) -> ChatCompletionUserMessage: + pass + + +def skip_empty_text_blocks( + message: ChatCompletionAssistantMessage | ChatCompletionUserMessage, +) -> ChatCompletionAssistantMessage | ChatCompletionUserMessage: + """ + Skips empty text blocks in message content text blocks. + + Do not insert content here. This is a helper function, which can also be used in base case. + """ + content_block: Final = message.get("content", None) + if content_block is None: + return message + if ( + isinstance(content_block, str) + and not content_block.strip() + and is_non_content_values_set(message) + and message["role"] == "assistant" + ): + modified_message = message.copy() + modified_message["content"] = None # user message content cannot be None + return modified_message + elif isinstance(content_block, list): + modified_content_block: Final = _skip_empty_dict_blocks(cast(list[dict], content_block)) + + # If no content remains and it's an assistant message, set content to None + if not modified_content_block and message["role"] == "assistant": + modified_message = message.copy() + modified_message["content"] = None + return modified_message + + modified_message_alt: Final = message.copy() + + # Type-specific casting based on message role + if message["role"] == "assistant": + modified_message_alt["content"] = cast( + list[OpenAIMessageContentListBlock] | None, + modified_content_block or None, + ) + elif message["role"] == "user" and modified_content_block is not None: + modified_message_alt["content"] = cast(list[ChatCompletionTextObject] | None, modified_content_block) + + return modified_message_alt + + return message + + +def process_empty_text_blocks( + message: ChatCompletionAssistantMessage, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + modified_content_block = message.get("content", None) + ## BASE CASE ## + if modified_content_block is None or not isinstance(modified_content_block, list): + return message + + # Check if all items are empty text blocks + if all(item["type"] == "text" and not item["text"].strip() for item in modified_content_block): + # Replace with a single continue message + _assistant_continue_message: Final = return_assistant_continue_message(assistant_continue_message) + modified_content_block = [ + { + "type": "text", + "text": convert_content_list_to_str(_assistant_continue_message), + } + ] + else: + # Filter out only empty text blocks, keeping non-empty text and other block types + modified_content_block = [ + item for item in modified_content_block if not (item["type"] == "text" and not item["text"].strip()) + ] + + modified_message: Final = message.copy() + modified_message["content"] = cast( + list[ChatCompletionTextObject] | list[ChatCompletionThinkingBlock], + modified_content_block, + ) + return modified_message + + +def get_assistant_message_block_or_continue_message( + message: ChatCompletionAssistantMessage, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + """ + Returns the user content block + if content block is an empty string, then return the default continue message + + Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 + """ + content_block: Final = message.get("content", None) + + # Handle Base case + if content_block is None or (assistant_continue_message is None and litellm.modify_params is False): + return skip_empty_text_blocks(message=message) + + # Handle string case + if isinstance(content_block, str): + # check if content is empty + if content_block.strip(): + return message + else: + if is_non_content_values_set(message): + modified_message: Final = message.copy() + modified_message["content"] = None + return modified_message + return return_assistant_continue_message(assistant_continue_message) + + # Handle list case + if isinstance(content_block, list): + """ + CHECK FOR + "content": [ + { + "type": "text", + "text": "" + } + ], + """ + return process_empty_text_blocks(message=message, assistant_continue_message=assistant_continue_message) + + # Handle unsupported type + raise ValueError(f"Unsupported content type: {type(content_block)}") + + +class BedrockConverseMessagesProcessor: + @staticmethod + def _initial_message_setup( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + ) -> list: + # gracefully handle base case of no messages at all + if len(messages) == 0: + if user_continue_message is not None: + messages.append(user_continue_message) + elif litellm.modify_params: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE) + else: + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + "bedrock requires at least one non-system message", + model=model, + llm_provider=llm_provider, + ) + + # if initial message is assistant message + if messages[0].get("role") is not None and messages[0]["role"] == "assistant": + if not messages[0].get("prefix"): + if user_continue_message is not None: + messages.insert(0, user_continue_message) + elif litellm.modify_params: + messages.insert(0, DEFAULT_USER_CONTINUE_MESSAGE) + + # if final message is assistant message + if messages[-1].get("role") is not None and messages[-1]["role"] == "assistant": + if not messages[-1].get("prefix"): + if user_continue_message is not None: + messages.append(user_continue_message) + elif litellm.modify_params: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE) + return messages + + @staticmethod + async def _bedrock_converse_messages_pt_async( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, + ) -> list[BedrockMessageBlock]: + contents: list[BedrockMessageBlock] = [] + msg_i = 0 + + messages = BedrockConverseMessagesProcessor._initial_message_setup( + messages, model, llm_provider, user_continue_message + ) + + while msg_i < len(messages): + user_content: list[BedrockContentBlock] = [] + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "user": + message_block = get_user_message_block_or_continue_message( + message=messages[msg_i], + user_continue_message=user_continue_message, + ) + if isinstance(message_block["content"], list): + _parts: list[BedrockContentBlock] = [] + for element in message_block["content"]: + if isinstance(element, dict): + if element["type"] == "text": + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "guarded_text": + # Wrap guarded_text in guardContent block + _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) + _parts.append(_part) + elif element["type"] in ("grounding_source", "query"): + # Contextual grounding tags are guardrail metadata; the + # model only needs the underlying text, so render them + # as plain text on the generate path. + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "image_url": + format: str | None = None + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + format = element["image_url"].get("format") + else: + image_url = element["image_url"] + _part = await BedrockImageProcessor.process_image_async( + image_url=image_url, format=format + ) + _parts.append(_part) + elif element["type"] == "file": + _part = await BedrockConverseMessagesProcessor._async_process_file_message( + message=cast(ChatCompletionFileObject, element) + ) + _parts.append(_part) + elif element["type"] == "document": + _part = BedrockConverseMessagesProcessor._process_document_message(element) + _parts.append(_part) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + user_content.extend(_parts) + elif message_block["content"] and isinstance(message_block["content"], str): + _part = BedrockContentBlock(text=messages[msg_i]["content"]) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + user_content.append(_part) + if _cache_point_block is not None: + user_content.append(_cache_point_block) + + msg_i += 1 + if user_content: + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=user_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(user_content) + else: + contents.append(BedrockMessageBlock(role="user", content=user_content)) + + ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## + tool_content: list[BedrockContentBlock] = [] + while msg_i < len(messages) and messages[msg_i]["role"] == "tool": + current_message = messages[msg_i] + tool_call_result = _convert_to_bedrock_tool_call_result(current_message) + tool_content.append(tool_call_result) + + # Check if we need to add a separate cachePoint block + tool_msg_cache_control = None + + # Check for message-level cache_control + if current_message.get("cache_control", None) is not None: + tool_msg_cache_control = current_message["cache_control"] + # Check for content-level cache_control in list content + elif isinstance(current_message.get("content"), list): + for content_element in current_message["content"]: + if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: + tool_msg_cache_control = content_element["cache_control"] + break + + # Add a separate cachePoint block if cache_control is present + if tool_msg_cache_control is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool_msg_cache_control}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + tool_content.append(cache_point_block) + + msg_i += 1 + # Deduplicate toolResult blocks with the same toolUseId + tool_content = _deduplicate_bedrock_tool_content(tool_content) + if tool_content: + # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(tool_content) + else: + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + assistant_content: list[BedrockContentBlock] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_message_block = get_assistant_message_block_or_continue_message( + message=messages[msg_i], + assistant_continue_message=assistant_continue_message, + ) + assistant_message_content_start: int = len(assistant_content) + _assistant_content = assistant_message_block.get("content", None) + thinking_blocks = cast( + list[ChatCompletionThinkingBlock] | None, + assistant_message_block.get("thinking_blocks"), + ) + + if thinking_blocks is not None: + converted_thinking_blocks = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, + ) + + if _assistant_content is not None and isinstance(_assistant_content, list): + assistants_parts: list[BedrockContentBlock] = [] + for element in _assistant_content: + if isinstance(element, dict): + if element["type"] == "thinking": + thinking_block = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] + ) + assistants_parts = ( + BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) + ) + elif element["type"] == "text": + # Skip completely empty strings to avoid blank content blocks + if element.get("text", "").strip(): + assistants_part = BedrockContentBlock(text=element["text"]) + assistants_parts.append(assistants_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + assistants_part = await BedrockImageProcessor.process_image_async(image_url=image_url) + assistants_parts.append(assistants_part) + # Add cache point block for assistant content elements + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + assistants_parts.append(_cache_point_block) + assistant_content.extend(assistants_parts) + elif _assistant_content is not None and isinstance(_assistant_content, str): + # Skip completely empty strings to avoid blank content blocks + if _assistant_content.strip(): + assistant_content.append(BedrockContentBlock(text=_assistant_content)) + # If content is empty/whitespace, skip it (don't add a placeholder) + # Add cache point block for assistant string content + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + + _tool_calls = assistant_message_block.get("tool_calls", []) + if _tool_calls: + assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + + msg_i += 1 + + assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") + assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) + + if assistant_content: + contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) + + @staticmethod + def translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks: list[ChatCompletionThinkingBlock], + ) -> list[BedrockContentBlock]: + reasoning_content_blocks: Final[list[BedrockContentBlock]] = [] + for thinking_block in thinking_blocks: + reasoning_text = thinking_block.get("thinking") + reasoning_signature = thinking_block.get("signature") + text_block = BedrockConverseReasoningTextBlock( + text=reasoning_text or "", + ) + if reasoning_signature is not None: + text_block["signature"] = reasoning_signature + reasoning_content_block = BedrockConverseReasoningContentBlock( + reasoningText=text_block, + ) + bedrock_content_block = BedrockContentBlock(reasoningContent=reasoning_content_block) + reasoning_content_blocks.append(bedrock_content_block) + return reasoning_content_blocks + + @staticmethod + def _process_file_message(message: ChatCompletionFileObject) -> BedrockContentBlock: + file_message: Final = message.get("file") + if file_message is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="bedrock", + ) + file_data: Final = file_message.get("file_data") + file_id: Final = file_message.get("file_id") + + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={message}", + model="", + llm_provider="bedrock", + ) + format: Final = file_message.get("format") + return BedrockImageProcessor.process_image_sync(image_url=cast(str, file_id or file_data), format=format) + + @staticmethod + async def _async_process_file_message( + message: ChatCompletionFileObject, + ) -> BedrockContentBlock: + file_message: Final = message.get("file") + if file_message is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="bedrock", + ) + file_data: Final = file_message.get("file_data") + file_id: Final = file_message.get("file_id") + format: Final = file_message.get("format") + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={message}", + model="", + llm_provider="bedrock", + ) + return await BedrockImageProcessor.process_image_async(image_url=cast(str, file_id or file_data), format=format) + + @staticmethod + def _process_document_message(element: dict) -> BedrockContentBlock: + """Convert a document content block to a Bedrock DocumentBlock. + + Handles the Anthropic-style document format: + {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "..."}} + """ + source: Final = element["source"] + source_type: Final = source.get("type") + if source_type != "base64": + raise ValueError( + f"Bedrock Converse only supports base64-encoded document sources, got '{source_type}'. " + "Please convert the document to base64 before sending to Bedrock." + ) + media_type: Final[str] = source["media_type"] + data: Final[str] = source["data"] + doc_format = BedrockImageProcessor._validate_format(mime_type=media_type, image_format=media_type.split("/")[1]) + + # Deterministic name using the same hashing pattern as _create_bedrock_block + HASH_SAMPLE_BYTES: Final = 64 * 1024 + normalized: Final = "".join(data.split()).encode("utf-8") + sample: Final = normalized[:HASH_SAMPLE_BYTES] + hasher: Final = hashlib.sha256() + hasher.update(sample) + hasher.update(str(len(normalized)).encode("utf-8")) + content_hash: Final = hasher.hexdigest()[:16] + document_name: Final = f"Document_{content_hash}_{doc_format}" + + return BedrockContentBlock( + document=BedrockDocumentBlock( + source=BedrockSourceBlock(bytes=data), + format=doc_format, + name=document_name, + ) + ) + + @staticmethod + def add_thinking_blocks_to_assistant_content( + thinking_blocks: list[BedrockContentBlock], + assistant_parts: list[BedrockContentBlock], + ) -> list[BedrockContentBlock]: + """ + If contains 'signature', it is a thinking block. + If missing 'signature', it is a text block - e.g. when using a non-anthropic model. + + Handle error raised by bedrock if thinking blocks are provided for a non-thinking model (e.g. nova with tool use) + + Relevant Issue: https://github.com/BerriAI/litellm/issues/9063 + """ + filtered_thinking_blocks: Final = [] + for block in thinking_blocks: + reasoning_content = block.get("reasoningContent", None) + reasoning_text = reasoning_content.get("reasoningText", None) if reasoning_content is not None else None + if reasoning_text and not reasoning_text.get("signature"): + reasoning_text_text = reasoning_text["text"] + if reasoning_text_text.strip(): + assistants_part = BedrockContentBlock(text=reasoning_text_text) + assistant_parts.append(assistants_part) + else: + filtered_thinking_blocks.append(block) + if len(filtered_thinking_blocks) > 0: + assistant_parts.extend(filtered_thinking_blocks) + return assistant_parts + + +def _bedrock_converse_messages_pt( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> list[BedrockMessageBlock]: + """ + Converts given messages from OpenAI format to Bedrock format + + - Roles must alternate b/w 'user' and 'model' (same as anthropic -> merge consecutive roles) + - Please ensure that function response turn comes immediately after a function call turn + - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 + """ + + contents: list[BedrockMessageBlock] = [] + msg_i = 0 + + messages = BedrockConverseMessagesProcessor._initial_message_setup( + messages, model, llm_provider, user_continue_message + ) + + while msg_i < len(messages): + user_content: list[BedrockContentBlock] = [] + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "user": + message_block = get_user_message_block_or_continue_message( + message=messages[msg_i], + user_continue_message=user_continue_message, + ) + if isinstance(message_block["content"], list): + _parts: list[BedrockContentBlock] = [] + for element in message_block["content"]: + if isinstance(element, dict): + if element["type"] == "text": + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "guarded_text": + # Wrap guarded_text in guardContent block + _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) + _parts.append(_part) + elif element["type"] in ("grounding_source", "query"): + # Contextual grounding tags are guardrail metadata; the + # model only needs the underlying text, so render them as + # plain text on the generate path. + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "image_url": + format: str | None = None + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + format = element["image_url"].get("format") + else: + image_url = element["image_url"] + _part = BedrockImageProcessor.process_image_sync( + image_url=image_url, + format=format, + ) + _parts.append(_part) + elif element["type"] == "file": + _part = BedrockConverseMessagesProcessor._process_file_message( + message=cast(ChatCompletionFileObject, element) + ) + _parts.append(_part) + elif element["type"] == "document": + _part = BedrockConverseMessagesProcessor._process_document_message(element) + _parts.append(_part) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + user_content.extend(_parts) + elif message_block["content"] and isinstance(message_block["content"], str): + _part = BedrockContentBlock(text=messages[msg_i]["content"]) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + user_content.append(_part) + if _cache_point_block is not None: + user_content.append(_cache_point_block) + + msg_i += 1 + if user_content: + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=user_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(user_content) + else: + contents.append(BedrockMessageBlock(role="user", content=user_content)) + + ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## + tool_content: list[BedrockContentBlock] = [] + while msg_i < len(messages) and messages[msg_i]["role"] == "tool": + tool_call_result = _convert_to_bedrock_tool_call_result(messages[msg_i]) + current_message = messages[msg_i] + + # Add the tool result first + tool_content.append(tool_call_result) + + # Check if we need to add a separate cachePoint block + tool_msg_cache_control = None + + # Check for message-level cache_control + if current_message.get("cache_control", None) is not None: + tool_msg_cache_control = current_message["cache_control"] + # Check for content-level cache_control in list content + elif isinstance(current_message.get("content"), list): + for content_element in current_message["content"]: + if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: + tool_msg_cache_control = content_element["cache_control"] + break + + # Add a separate cachePoint block if cache_control is present + if tool_msg_cache_control is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool_msg_cache_control}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + tool_content.append(cache_point_block) + + msg_i += 1 + # Deduplicate toolResult blocks with the same toolUseId + tool_content = _deduplicate_bedrock_tool_content(tool_content) + if tool_content: + # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(tool_content) + else: + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + assistant_content: list[BedrockContentBlock] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_message_block = get_assistant_message_block_or_continue_message( + message=messages[msg_i], + assistant_continue_message=assistant_continue_message, + ) + assistant_message_content_start: int = len(assistant_content) + _assistant_content = assistant_message_block.get("content", None) + thinking_blocks = cast( + list[ChatCompletionThinkingBlock] | None, + assistant_message_block.get("thinking_blocks"), + ) + + if thinking_blocks is not None: + converted_thinking_blocks = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, + ) + + if _assistant_content is not None and isinstance(_assistant_content, list): + assistants_parts: list[BedrockContentBlock] = [] + for element in _assistant_content: + if isinstance(element, dict): + if element["type"] == "thinking": + thinking_block = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] + ) + ) + assistants_parts = ( + BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) + ) + elif element["type"] == "text": + # AWS Bedrock doesn't allow empty or whitespace-only text content + # Skip completely empty strings to avoid blank content blocks + if element.get("text", "").strip(): + assistants_part = BedrockContentBlock(text=element["text"]) + assistants_parts.append(assistants_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + assistants_part = BedrockImageProcessor.process_image_sync(image_url=image_url) + assistants_parts.append(assistants_part) + # Add cache point block for assistant content elements + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + assistants_parts.append(_cache_point_block) + assistant_content.extend(assistants_parts) + elif _assistant_content is not None and isinstance(_assistant_content, str): + # Skip completely empty strings to avoid blank content blocks + if _assistant_content.strip(): + assistant_content.append(BedrockContentBlock(text=_assistant_content)) + # Add cache point block for assistant string content + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + _tool_calls = assistant_message_block.get("tool_calls", []) + if _tool_calls: + assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + assistant_content.append(_cache_point_block) + + msg_i += 1 + + assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") + assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) + + if assistant_content: + contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) + + +def make_valid_bedrock_tool_name(input_tool_name: str) -> str: + """Normalize tool names to Bedrock pattern [a-zA-Z][a-zA-Z0-9_-]*.""" + + def replace_invalid(char): + if char.isalnum() or char in ("_", "-"): + return char + return "_" + + # If the string is empty, return a default valid identifier + if input_tool_name is None or len(input_tool_name) == 0: + return input_tool_name + bedrock_tool_name = copy.copy(input_tool_name) + # If it doesn't start with a letter, prepend 'a' + if not bedrock_tool_name[0].isalpha(): + bedrock_tool_name = "a" + bedrock_tool_name + + # Replace any invalid characters with underscores + valid_string: Final = "".join(replace_invalid(char) for char in bedrock_tool_name) + + if input_tool_name != valid_string: + # passed tool name was formatted to become valid + # store it internally so we can use for the response + litellm.bedrock_tool_name_mappings.set_cache(key=valid_string, value=input_tool_name) + + return valid_string + + +def add_cache_point_tool_block(tool: dict, model: str | None = None) -> BedrockToolBlock | None: + from litellm.llms.bedrock.common_utils import ( + bedrock_model_accepts_cache_points, + is_claude_4_5_on_bedrock, + ) + + cache_control: Final = tool.get("cache_control", None) + if cache_control is not None and bedrock_model_accepts_cache_points(model): + cache_point: Final = cache_control.get("type", "ephemeral") + if cache_point == "ephemeral": + cache_point_block: Final[CachePointBlock] = {"type": "default"} + if isinstance(cache_control, dict) and "ttl" in cache_control: + ttl: Final = cache_control["ttl"] + if ttl in ["5m", "1h"] and model is not None and is_claude_4_5_on_bedrock(model): + cache_point_block["ttl"] = ttl + return {"cachePoint": cache_point_block} + return None + + +def _is_bedrock_tool_block(tool: dict) -> bool: + """ + Check if a tool is already a BedrockToolBlock. + + BedrockToolBlock has one of: systemTool, toolSpec, or cachePoint. + This is used to detect tools that are already in Bedrock format + (e.g., systemTool for Nova grounding) vs OpenAI-style function tools + that need transformation. + + Args: + tool: The tool dict to check + + Returns: + True if the tool is already a BedrockToolBlock, False otherwise + + Examples: + >>> _is_bedrock_tool_block({"systemTool": {"name": "nova_grounding"}}) + True + >>> _is_bedrock_tool_block({"type": "function", "function": {...}}) + False + """ + return isinstance(tool, dict) and ("systemTool" in tool or "toolSpec" in tool or "cachePoint" in tool) + + +def _bedrock_tools_pt(tools: list, model: str | None = None) -> list[BedrockToolBlock]: + """ + OpenAI tools looks like: + tools = [ + { + "type": "function", + "function": { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + } + } + ] + """ + """ + Bedrock toolConfig looks like: + "tools": [ + { + "toolSpec": { + "name": "top_song", + "description": "Get the most popular song played on a radio station.", + "inputSchema": { + "json": { + "type": "object", + "properties": { + "sign": { + "type": "string", + "description": "The call sign for the radio station for which you want the most popular song. Example calls signs are WZPZ, and WKRP." + } + }, + "required": [ + "sign" + ] + } + } + } + } + ] + """ + from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs + from litellm.llms.bedrock.common_utils import ( + bedrock_converse_supports_strict_tools, + normalize_json_schema_custom_types_to_object, + ) + + _valid_json_schema_root_types = frozenset(("array", "boolean", "integer", "null", "number", "object", "string")) + # Only Claude on Bedrock honours strict tool schemas; other families + # (Nova, Llama, GPT-OSS) reject the strict field outright. Opus 4.7/4.8 + # also reject `strict` on Bedrock Converse (see #31582) — their validator + # maps toolSpec to the native Anthropic tool shape, which has no strict + # field, even though Anthropic's native API accepts it as a top-level key. + supports_strict_tools: Final = bool(model and bedrock_converse_supports_strict_tools(model)) + tool_block_list: Final[list[BedrockToolBlock]] = [] + for tool_idx, tool in enumerate(tools): + # Check if tool is already a BedrockToolBlock (e.g., systemTool for Nova grounding) + if _is_bedrock_tool_block(tool): + # Already a BedrockToolBlock, pass it through + tool_block_list.append(tool) + continue + + # Responses built-in tools (web_search, image_generation, namespace, tool_search, + # custom) carry neither an OpenAI "function" nor an Anthropic "input_schema" and have + # no Bedrock toolSpec equivalent; drop them instead of emitting an empty junk toolSpec. + if isinstance(tool, dict) and "function" not in tool and "input_schema" not in tool: + continue + + # OpenAI function tools, or Anthropic Messages / Claude Code ({name, input_schema, type, ...}) + if isinstance(tool, dict) and "input_schema" in tool and "function" not in tool: + parameters = copy.deepcopy(tool.get("input_schema") or {"type": "object", "properties": {}}) + raw_name = tool.get("name", "") or "" + _tool_description = tool.get("description", None) + else: + parameters = copy.deepcopy(tool.get("function", {}).get("parameters", {"type": "object", "properties": {}})) + raw_name = tool.get("function", {}).get("name", "") or "" + _tool_description = tool.get("function", {}).get("description", None) + + if not (raw_name and str(raw_name).strip()): + raw_name = f"litellm_unnamed_tool_{tool_idx}" + + # related issue: https://github.com/BerriAI/litellm/issues/5007 + # Bedrock tool names must satisfy pattern: [a-zA-Z][a-zA-Z0-9_-]* + name = make_valid_bedrock_tool_name(input_tool_name=raw_name) + if _tool_description: # bedrock doesn't accept empty "" or None descriptions + description = _tool_description + else: + description = name + + defs = parameters.pop("$defs", {}) + defs_copy = copy.deepcopy(defs) + # Expand $ref references in parameters using the definitions + # Note: We don't pre-flatten defs as that causes exponential memory growth + # with circular references (see issue #19098). unpack_defs handles nested + # refs recursively and correctly detects/skips circular references. + unpack_defs(parameters, defs_copy) + normalize_json_schema_custom_types_to_object(parameters) + if parameters.get("type") not in _valid_json_schema_root_types: + parameters["type"] = "object" + tool_block = cast( + BedrockToolBlock, + BedrockToolSpec( + name=name, + description=description, + parameters=parameters, + strict=tool.get("function", {}).get("strict", None), + supports_strict_tools=supports_strict_tools, + ), + ) + tool_block_list.append(tool_block) + + ## ADD CACHE POINT TOOL BLOCK ## + cache_point_tool_block = add_cache_point_tool_block(tool, model=model) + if cache_point_tool_block is not None: + tool_block_list.append(cache_point_tool_block) + + return tool_block_list + + +# Function call template +def function_call_prompt(messages: list, functions: list): + function_prompt = """Produce JSON OUTPUT ONLY! Adhere to this format {"name": "function_name", "arguments":{"argument_name": "argument_value"}} The following functions are available to you:""" + for function in functions: + function_prompt += f"""\n{function}\n""" + + function_added_to_prompt = False + for message in messages: + if "system" in message["role"]: + if isinstance(message["content"], str): + message["content"] += f""" {function_prompt}""" + else: + message["content"].append({"type": "text", "text": f""" {function_prompt}"""}) + function_added_to_prompt = True + + if function_added_to_prompt is False: + messages.append({"role": "system", "content": f"""{function_prompt}"""}) + + return messages + + +def response_schema_prompt(model: str, response_schema: dict) -> str: + """ + Decides if a user-defined custom prompt or default needs to be used + + Returns the prompt str that's passed to the model as a user message + """ + custom_prompt_details: dict | None = None + response_schema_as_message: Final = [{"role": "user", "content": f"{response_schema}"}] + if f"{model}/response_schema_prompt" in litellm.custom_prompt_dict: + custom_prompt_details = litellm.custom_prompt_dict[ + f"{model}/response_schema_prompt" + ] # allow user to define custom response schema prompt by model + elif "response_schema_prompt" in litellm.custom_prompt_dict: + custom_prompt_details = litellm.custom_prompt_dict["response_schema_prompt"] + + if custom_prompt_details is not None: + return custom_prompt( + role_dict=custom_prompt_details["roles"], + initial_prompt_value=custom_prompt_details["initial_prompt_value"], + final_prompt_value=custom_prompt_details["final_prompt_value"], + messages=response_schema_as_message, + ) + else: + return default_response_schema_prompt(response_schema=response_schema) + + +def default_response_schema_prompt(response_schema: dict) -> str: + """ + Used if provider/model doesn't support 'response_schema' param. + + This is the default prompt. Allow user to override this with a custom_prompt. + """ + prompt_str: Final = f"""Use this JSON schema: + ```json + {response_schema} + ```""" + return prompt_str + + +# Custom prompt template +def custom_prompt( + role_dict: dict, + messages: list, + initial_prompt_value: str = "", + final_prompt_value: str = "", + bos_token: str = "", + eos_token: str = "", +) -> str: + prompt = bos_token + initial_prompt_value + bos_open = True + ## a bos token is at the start of a system / human message + ## an eos token is at the end of the assistant response to the message + for message in messages: + role = message["role"] + + if role in ["system", "human"] and not bos_open: + prompt += bos_token + bos_open = True + + pre_message_str = ( + role_dict[role]["pre_message"] if role in role_dict and "pre_message" in role_dict[role] else "" + ) + post_message_str = ( + role_dict[role]["post_message"] if role in role_dict and "post_message" in role_dict[role] else "" + ) + if isinstance(message["content"], str): + prompt += pre_message_str + message["content"] + post_message_str + elif isinstance(message["content"], list): + text_str = "" + for content in message["content"]: + if content.get("text", None) is not None and isinstance(content["text"], str): + text_str += content["text"] + prompt += pre_message_str + text_str + post_message_str + + if role == "assistant": + prompt += eos_token + bos_open = False + + prompt += final_prompt_value + return prompt + + +def prompt_factory( + model: str, + messages: list, + custom_llm_provider: str | None = None, + api_key: str | None = None, +): + original_model_name: Final = model + model = model.lower() + if custom_llm_provider == "ollama": + return ollama_pt(model=model, messages=messages) + elif custom_llm_provider == "anthropic": + if litellm.AnthropicTextConfig._is_anthropic_text_model(model): + return anthropic_pt(messages=messages) + return anthropic_messages_pt(messages=messages, model=model, llm_provider=custom_llm_provider) + elif custom_llm_provider == "anthropic_xml": + return anthropic_messages_pt_xml(messages=messages) + elif custom_llm_provider == "gemini": + if ( + model == "gemini-pro-vision" + or litellm.supports_vision(model=model) + or litellm.supports_vision(model=custom_llm_provider + "/" + model) + ): + return _gemini_vision_convert_messages(messages=messages) + else: + return gemini_text_image_pt(messages=messages) + elif custom_llm_provider == "mistral": + return litellm.MistralConfig()._transform_messages(messages=messages, model=model) + elif custom_llm_provider == "bedrock": + if "amazon.titan-text" in model: + return amazon_titan_pt(messages=messages) + elif "anthropic." in model: + if any(_ in model for _ in ["claude-2.1", "claude-v2:1"]): + return claude_2_1_pt(messages=messages) + else: + return anthropic_pt(messages=messages) + elif "mistral." in model: + return mistral_instruct_pt(messages=messages) + elif "llama2" in model and "chat" in model: + return llama_2_chat_pt(messages=messages) + elif ("llama3" in model or "llama4" in model) and "instruct" in model: + return hf_chat_template( + model="meta-llama/Meta-Llama-3-8B-Instruct", + messages=messages, + ) + + elif custom_llm_provider == "clarifai": + if "claude" in model: + return anthropic_pt(messages=messages) + + elif custom_llm_provider == "perplexity": + for message in messages: + message.pop("name", None) + return messages + elif custom_llm_provider == "azure_text": + return azure_text_pt(messages=messages) + elif custom_llm_provider == "watsonx": + from litellm.llms.watsonx.chat.transformation import IBMWatsonXChatConfig + + return IBMWatsonXChatConfig.apply_prompt_template(model=model, messages=messages) + + try: + if "meta-llama/llama-2" in model and "chat" in model: + return llama_2_chat_pt(messages=messages) + elif ("meta-llama/llama-3" in model or "meta-llama-3" in model) and "instruct" in model: + return hf_chat_template( + model="meta-llama/Meta-Llama-3-8B-Instruct", + messages=messages, + ) + elif ( + "tiiuae/falcon" in model + ): # Note: for the instruct models, it's best to use a User: .., Assistant:.. approach in your prompt template. + if model == "tiiuae/falcon-180B-chat": + return falcon_chat_pt(messages=messages) + elif "instruct" in model: + return falcon_instruct_pt(messages=messages) + elif "mosaicml/mpt" in model: + if "chat" in model: + return mpt_chat_pt(messages=messages) + elif "codellama/codellama" in model or "togethercomputer/codellama" in model: + if "instruct" in model: + return llama_2_chat_pt( + messages=messages + ) # https://huggingface.co/blog/codellama#conversational-instructions + elif "wizardlm/wizardcoder" in model: + return wizardcoder_pt(messages=messages) + elif "phind/phind-codellama" in model: + return phind_codellama_pt(messages=messages) + elif "togethercomputer/llama-2" in model and ("instruct" in model or "chat" in model): + return llama_2_chat_pt(messages=messages) + elif model in [ + "gryphe/mythomax-l2-13b", + "gryphe/mythomix-l2-13b", + "gryphe/mythologic-l2-13b", + ]: + return alpaca_pt(messages=messages) + elif model.strip() == "meta-llama/llamaguard-7b": + chat_template = "{% if messages|length % 2 == 0 %}{% set role = 'Agent' %}{% else %}{% set role = 'User' %}{% endif %}{{ \"[INST] Task: Check if there is unsafe content in '\" + role + \"' messages in conversations according our safety policy with the below categories.\n\n\nO1: Violence and Hate. \nShould not\n- Help people plan or engage in violence.\n- Encourage people to engage in violence.\n- Express hateful, derogatory, or demeaning sentiments against people based on sensitive personal characteristics like their race, color, religion, national origin, sexual orientation, gender, gender identity, or disability.\n- Encourage discrimination against people based on those characteristics and should not use slurs or create insults based on those characteristics.\nCan\n- Provide information on violence and discrimination.\n- Discuss the topics of hate and violence, and can discuss historical events involving violence.\nO2: Sexual Content.\nShould not\n- Engage in sexually explicit (i.e., erotic) conversations and should not encourage people to engage in sexual activities.\nCan\n- Discuss topics related to sexuality and sexual education.\n- Provide general information regarding sexual activities.\n- Provide guidance on topics related to sexual health.\nO3: Criminal Planning.\nShould not\n- Help people plan or engage in criminal activities like theft, kidnapping, or financial crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in criminal activities or condone criminal behavior.\nCan\n- Help people understand how crimes are committed, provided that the response is not so detailed that it could be used to provide a roadmap for future crimes.\n- Help people understand how systems work that are used to prevent crimes, provided those descriptions do not provide detailed information on how to defeat those systems.\nO4: Guns and Illegal Weapons.\nShould not\n- Assist in the planning or execution of firearm or illegal weapons crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in firearm or illegal weapons crimes or condone such behavior.\nCan\n- Discuss firearms and the arguments for and against firearm ownership.\nO5: Regulated or Controlled Substances.\nShould not\n- Encourage or assist people in the trafficking, consumption, or use of regulated or controlled substances, including but not limited to illegal drugs, tobacco, alcohol, and cannabis.\n- Assist or encourage people to create such substances when it is illegal to do so.\nCan\n- Discuss the use of regulated or controlled substances, and can provide general information on those substances (ex: what their chemical composition is and what their effects are).\n- Provide information on how those substances are manufactured, provided that the information is not so detailed as to provide step-by-step instructions for their creation.\nO6: Self-Harm.\nShould not\n- Encourage people to harm themselves.\n- Romanticize or condone self-harm or suicide.\n- Provide information regarding the methods of suicide or self-harm.\n- Provide any information or advice, other than information regarding appropriate health resources, to people who express or imply that they wish to harm themselves.\nCan\n- Provide general information regarding self-harm and suicide (ex: statistics, the mental health challenges faced by those who take these actions).\nShould\n- Provide information regarding appropriate health resources to people who imply or express the desire to harm themselves.\n\n\n\n\n\" }}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{% set role = 'User' %}{% elif message['role'] == 'assistant' %}{% set role = 'Agent' %}{% endif %}{{ role + ': ' + content.strip() + '\n\n' }}{% endfor %}{{ \"\n\nProvide your safety assessment for \" + role + \" in the above conversation:\n- First line must read 'safe' or 'unsafe'.\n- If unsafe, a second line must include a comma-separated list of violated categories. [/INST]\" }}" + return hf_chat_template(model=model, messages=messages, chat_template=chat_template) + else: + return hf_chat_template(original_model_name, messages) + except Exception: + return default_pt( + messages=messages + ) # default that covers Bloom, T-5, any non-chat tuned model (e.g. base Llama2) + + +def get_attribute_or_key(tool_or_function, attribute, default=None): + if hasattr(tool_or_function, attribute): + return getattr(tool_or_function, attribute) + if isinstance(tool_or_function, Mapping): + return tool_or_function.get(attribute, default) + return default + + +class NormalizedToolCall(TypedDict): + id: str | None + name: str | None + arguments: dict[str, object] + + +_ArgumentObjects: TypeAlias = tuple[dict[str, object], ...] +_ParsedToolCall: TypeAlias = tuple[str | None, str | None, _ArgumentObjects] + + +def _optional_call_id(value: object) -> str | None: + if isinstance(value, str) and value: + return value + return None + + +def _optional_tool_name(value: object) -> str | None: + if isinstance(value, str): + return value + return None + + +def _split_tool_call_ids(calls: Sequence[tuple[str | None, int]]) -> tuple[tuple[str | None, ...], ...]: + taken: Final = frozenset(_sanitize_anthropic_tool_use_id(call_id) for call_id, _ in calls if call_id) + + def fresh(call_id: str) -> Iterator[str]: + return filter( + lambda candidate: _sanitize_anthropic_tool_use_id(candidate) not in taken, + (f"{call_id}__concat_{n}" for n in itertools.count(1)), + ) + + suffixes: Final = MappingProxyType( + {_sanitize_anthropic_tool_use_id(call_id): fresh(call_id) for call_id, count in calls if call_id and count > 1} + ) + return tuple( + ( + call_id, + *(next(suffixes[_sanitize_anthropic_tool_use_id(call_id)]) for _ in range(count - 1)), + ) + if call_id + else (None,) * count + for call_id, count in calls + ) + + +def _parse_tool_call_arguments(raw: object, tool_name: str | None, context: str) -> _ArgumentObjects: + # Anthropic's tool_use blocks already carry a parsed dict in "input"; + # chat completions and the Responses API carry a JSON string that may be + # truncated by the model, so route those through the repair-aware parser. + if isinstance(raw, dict): + return (raw,) + if not isinstance(raw, str): + return ({},) + normalized_raw: Final = "{}" if raw == REDACTED_BY_LITELLM else raw + try: + parsed: Final = parse_tool_call_arguments(normalized_raw, tool_name=tool_name, context=context) + except ValueError as e: + salvaged: Final = salvage_concatenated_tool_arguments(normalized_raw) + if salvaged: + verbose_logger.warning( + "Recovered %d tool call(s) from concatenated JSON arguments for tool '%s' (%s)", + len(salvaged), + tool_name or "", + context, + ) + return salvaged + verbose_logger.warning("Failed to parse tool call arguments: %s", e) + return ({},) + return (parsed,) if isinstance(parsed, dict) else ({},) + + +def _choice_tool_calls(choice: object) -> tuple[object, ...]: + message: Final = get_attribute_or_key(choice, "message", None) + tool_calls: Final = get_attribute_or_key(message, "tool_calls", None) if message is not None else None + if isinstance(tool_calls, list): + return tuple(tool_calls) + return () + + +def _selected_choices(response: object, include_all_choices: bool) -> tuple[object, ...]: + choices: Final = get_attribute_or_key(response, "choices", None) + if not isinstance(choices, list) or not choices: + return () + if include_all_choices: + return tuple(choices) + return (choices[0],) + + +def _parsed_chat_tool_call(tool_call: object) -> _ParsedToolCall | None: + function: Final = get_attribute_or_key(tool_call, "function", None) + if function is None: + return None + name: Final = _optional_tool_name(get_attribute_or_key(function, "name")) + return ( + _optional_call_id(get_attribute_or_key(tool_call, "id")), + name, + _parse_tool_call_arguments( + get_attribute_or_key(function, "arguments", "{}"), + tool_name=name, + context="chat completions", + ), + ) + + +def _parsed_calls_in_choice(choice: object) -> tuple[_ParsedToolCall, ...]: + return tuple( + parsed for tool_call in _choice_tool_calls(choice) if (parsed := _parsed_chat_tool_call(tool_call)) is not None + ) + + +def _parsed_chat_tool_calls(response: object, include_all_choices: bool) -> tuple[_ParsedToolCall, ...]: + grouped: Final = tuple( + _parsed_calls_in_choice(choice) for choice in _selected_choices(response, include_all_choices) + ) + return tuple(itertools.chain.from_iterable(grouped)) + + +def _normalized_tool_calls_for_parse( + name: str | None, + call_ids: tuple[str | None, ...], + arguments: _ArgumentObjects, +) -> tuple[NormalizedToolCall, ...]: + return tuple( + NormalizedToolCall(id=call_id, name=name, arguments=argument) + for call_id, argument in zip(call_ids, arguments, strict=True) + ) + + +def _normalized_tool_calls_from_parses(parses: Sequence[_ParsedToolCall]) -> tuple[NormalizedToolCall, ...]: + id_groups: Final = _split_tool_call_ids(tuple((call_id, len(arguments)) for call_id, _, arguments in parses)) + grouped: Final = tuple( + _normalized_tool_calls_for_parse(name, call_ids, arguments) + for (_, name, arguments), call_ids in zip(parses, id_groups, strict=True) + ) + return tuple(itertools.chain.from_iterable(grouped)) + + +def _tool_calls_from_chat_completion_response( + response: object, include_all_choices: bool = False +) -> tuple[NormalizedToolCall, ...]: + return _normalized_tool_calls_from_parses(_parsed_chat_tool_calls(response, include_all_choices)) + + +def _response_function_calls(response: object) -> tuple[object, ...]: + output: Final = get_attribute_or_key(response, "output", None) + if not isinstance(output, list): + return () + return tuple(item for item in output if get_attribute_or_key(item, "type") == "function_call") + + +def _parsed_response_tool_call(item: object) -> _ParsedToolCall: + name: Final = _optional_tool_name(get_attribute_or_key(item, "name")) + raw_id: Final = get_attribute_or_key(item, "call_id") or get_attribute_or_key(item, "id") + return ( + _optional_call_id(raw_id), + name, + _parse_tool_call_arguments( + get_attribute_or_key(item, "arguments", "{}"), + tool_name=name, + context="responses API", + ), + ) + + +def _tool_calls_from_responses_api_response(response: object) -> tuple[NormalizedToolCall, ...]: + parses: Final = tuple(_parsed_response_tool_call(item) for item in _response_function_calls(response)) + return _normalized_tool_calls_from_parses(parses) + + +def _tool_calls_from_anthropic_messages_response(response: object) -> list[NormalizedToolCall]: + content: Final = get_attribute_or_key(response, "content", None) + if not isinstance(content, list): + return [] + result: Final[list[NormalizedToolCall]] = [] + for block in content: + if get_attribute_or_key(block, "type") != "tool_use": + continue + raw_input = get_attribute_or_key(block, "input", {}) + result.append( + NormalizedToolCall( + id=get_attribute_or_key(block, "id"), + name=get_attribute_or_key(block, "name"), + arguments=raw_input if isinstance(raw_input, dict) else {}, + ) + ) + return result + + +def get_tool_calls_from_response(response: object, include_all_choices: bool = False) -> list[NormalizedToolCall]: + """ + Extract tool/function calls from a response object into a normalized + ``{"id", "name", "arguments"}`` shape, regardless of which API surface + produced it: chat completions (``choices[].message.tool_calls``), + the Responses API (``output`` items of type ``function_call``), or the + Anthropic Messages API (``content`` blocks of type ``tool_use``). + + ``include_all_choices`` decides the chat-completions scope: the default + reads only ``choices[0]``, which is what consumers that act on THE reply + (e.g. guardrails rebuilding the primary assistant message) want; usage + accounting passes True because every choice of an ``n>1`` request costs + money and its tool calls really ran. The other surfaces have a single + output, so the flag has no effect on them. + + Callers that only care about a specific tool should filter the result by + ``name`` themselves -- this returns every tool call found. + """ + chat_tool_calls: Final = _tool_calls_from_chat_completion_response( + response, include_all_choices=include_all_choices + ) + if chat_tool_calls: + return list(chat_tool_calls) + for extractor in ( + _tool_calls_from_responses_api_response, + _tool_calls_from_anthropic_messages_response, + ): + tool_calls = extractor(response) + if tool_calls: + return list(tool_calls) + return [] + + +def has_tool_with_name(tools: object, tool_name: str) -> bool: + """ + Check whether a tools list (as sent to an LLM) includes a tool with the + given name, regardless of shape: OpenAI-style function tools + (``{"type": "function", "function": {"name": ...}}``) or Anthropic's + native tool shape (a top-level ``"name"``, e.g. + ``{"name": ..., "input_schema": ...}``). Anthropic's documented client + tool format doesn't require a ``"type"`` key at all -- ``"custom"`` is + only one of several possible values -- so any non-OpenAI-shaped tool is + matched on its top-level ``"name"``. + """ + if not isinstance(tools, list): + return False + for tool in tools: + if not isinstance(tool, dict): + continue + function = tool.get("function") + if tool.get("type") == "function" and isinstance(function, dict): + if function.get("name") == tool_name: + return True + elif tool.get("name") == tool_name: + return True + return False + + +def resolve_structured_messages( + messages: list[dict[str, object]] | None, + request_kwargs: dict[str, Any], +) -> list[dict[str, object]] | None: + """ + Normalize a request's messages to OpenAI-spec chat-completions shape, + regardless of which API surface produced them (chat completions, + Anthropic /v1/messages, Responses API ``input``, etc). + + Returns ``messages`` unchanged if already present. Otherwise dispatches + through the guardrail translation handlers (the same per-surface + conversion logic guardrails use) to convert e.g. Responses API ``input`` + into a message list. Returns ``None`` if no messages could be resolved. + """ + if messages: + return messages + + from litellm.litellm_core_utils.api_route_to_call_types import ( + get_call_types_for_route, + ) + from litellm.llms import load_guardrail_translation_mappings + from litellm.types.utils import CallTypes + + mappings: Final = load_guardrail_translation_mappings() + call_type: CallTypes | None = None + + # 1. Try route-based inference from proxy metadata + route: Final = request_kwargs.get("litellm_metadata", {}).get("user_api_key_request_route") + if route: + call_types_list: Final = get_call_types_for_route(route) + if call_types_list: + for ct in call_types_list: + if ct in mappings: + call_type = ct + break + + # 2. Fallback: try each mapped handler until one produces messages + handlers_to_try: Final[list[Any]] = [] + if call_type is not None and call_type in mappings: + handlers_to_try.append(mappings[call_type]()) + else: + handlers_to_try.extend(handler_cls() for handler_cls in mappings.values()) + + for handler in handlers_to_try: + structured = handler.get_structured_messages(request_kwargs) + if structured: + return [msg if isinstance(msg, dict) else msg.model_dump() for msg in structured] + return None From 2ca368d27db3732a9798ac61bdaea6457ef8505c Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Mon, 28 Sep 2026 00:38:43 +0800 Subject: [PATCH 5/8] fix(bedrock): preserve assistant cache point boundaries --- .../prompt_templates/factory.py | 11530 ++++++++-------- litellm/llms/anthropic/chat/transformation.py | 2 +- ...llm_core_utils_prompt_templates_factory.py | 27 + 3 files changed, 5800 insertions(+), 5759 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 6d7c3edd723..9b3cad5db2a 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -1,5758 +1,5772 @@ -import base64 -import copy -import hashlib -import itertools -import json -import mimetypes -import re -import xml.etree.ElementTree as ET -from collections.abc import Iterator, Mapping, Sequence -from enum import Enum -from types import MappingProxyType -from typing import Any, Final, TypeAlias, TypedDict, cast, overload - -from jinja2.sandbox import ImmutableSandboxedEnvironment - -import litellm -import litellm.types -import litellm.types.llms -from litellm import verbose_logger -from litellm._uuid import uuid -from litellm.constants import REDACTED_BY_LITELLM -from litellm.litellm_core_utils.prompt_templates.mid_conversation_system import anthropic_system_messages -from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get -from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client -from litellm.types.files import get_file_extension_from_mime_type -from litellm.types.llms.anthropic import * -from litellm.types.llms.bedrock import CachePointBlock -from litellm.types.llms.bedrock import MessageBlock as BedrockMessageBlock -from litellm.types.llms.custom_http import httpxSpecialProvider -from litellm.types.llms.ollama import OllamaVisionModelObject -from litellm.types.llms.openai import ( - AllMessageValues, - ChatCompletionAssistantMessage, - ChatCompletionAssistantToolCall, - ChatCompletionFileObject, - ChatCompletionFileObjectFile, - ChatCompletionFunctionMessage, - ChatCompletionImageObject, - ChatCompletionImageUrlObject, - ChatCompletionTextObject, - ChatCompletionToolCallFunctionChunk, - ChatCompletionToolMessage, - ChatCompletionUserMessage, - OpenAIMessageContentListBlock, -) -from litellm.types.llms.vertex_ai import FunctionCall as VertexFunctionCall -from litellm.types.llms.vertex_ai import FunctionResponse as VertexFunctionResponse -from litellm.types.llms.vertex_ai import PartType as VertexPartType -from litellm.types.utils import GenericImageParsingChunk - -from .common_utils import ( - convert_content_list_to_str, - infer_content_type_from_url_and_content, - is_non_content_values_set, - is_unsignable_thinking_block, - parse_tool_call_arguments, - salvage_concatenated_tool_arguments, -) -from .image_handling import convert_url_to_base64 - - -def default_pt(messages): - return " ".join(message["content"] for message in messages) - - -def prompt_injection_detection_default_pt(): - return """Detect if a prompt is safe to run. Return 'UNSAFE' if not.""" - - -BAD_MESSAGE_ERROR_STR: Final = "Invalid Message " - -# Separator used to embed Gemini thought signatures in tool call IDs -# See: https://ai.google.dev/gemini-api/docs/thought-signatures -THOUGHT_SIGNATURE_SEPARATOR: Final = "__thought__" - -# used to interweave user messages, to ensure user/assistant alternating -DEFAULT_USER_CONTINUE_MESSAGE: Final = { - "role": "user", - "content": "Please continue.", -} # similar to autogen. Only used if `litellm.modify_params=True`. - -DEFAULT_USER_CONTINUE_MESSAGE_TYPED: Final = ChatCompletionUserMessage( - role="user", - content="Please continue.", -) - -# used to interweave assistant messages, to ensure user/assistant alternating -DEFAULT_ASSISTANT_CONTINUE_MESSAGE: Final = ChatCompletionAssistantMessage( - role="assistant", - content=[ - { - "type": "text", - "text": "Please continue.", - } - ], -) # similar to autogen. Only used if `litellm.modify_params=True`. - - -def map_system_message_pt(messages: list) -> list: - """ - Convert 'system' message to 'user' message if provider doesn't support 'system' role. - - Enabled via `completion(...,supports_system_message=False)` - - If next message is a user message or assistant message -> merge system prompt into it - - if next message is system -> append a user message instead of the system message - """ - - new_messages: Final = [] - for i, m in enumerate(messages): - if m["role"] == "system": - if i < len(messages) - 1: # Not the last message - next_m = messages[i + 1] - next_role = next_m["role"] - if next_role == "user" or next_role == "assistant": # Next message is a user or assistant message - # Merge system prompt into the next message - next_m["content"] = m["content"] + " " + next_m["content"] - elif next_role == "system": # Next message is a system message - # Append a user message instead of the system message - new_message = {"role": "user", "content": m["content"]} - new_messages.append(new_message) - else: # Last message - new_message = {"role": "user", "content": m["content"]} - new_messages.append(new_message) - else: # Not a system message - new_messages.append(m) - - return new_messages - - -# alpaca prompt template - for models like mythomax, etc. -def alpaca_pt(messages): - prompt: Final = custom_prompt( - role_dict={ - "system": { - "pre_message": "### Instruction:\n", - "post_message": "\n\n", - }, - "user": { - "pre_message": "### Instruction:\n", - "post_message": "\n\n", - }, - "assistant": {"pre_message": "### Response:\n", "post_message": "\n\n"}, - }, - bos_token="", - eos_token="", - messages=messages, - ) - return prompt - - -# Llama2 prompt template -def llama_2_chat_pt(messages): - prompt: Final = custom_prompt( - role_dict={ - "system": { - "pre_message": "[INST] <>\n", - "post_message": "\n<>\n [/INST]\n", - }, - "user": { # follow this format https://github.com/facebookresearch/llama/blob/77062717054710e352a99add63d160274ce670c6/llama/generation.py#L348 - "pre_message": "[INST] ", - "post_message": " [/INST]\n", - }, - "assistant": { - "post_message": "\n" # follows this - https://replicate.com/blog/how-to-prompt-llama - }, - }, - messages=messages, - bos_token="", - eos_token="", - ) - return prompt - - -def convert_to_ollama_image(openai_image_url: str): - try: - if openai_image_url.startswith("http"): - openai_image_url = convert_url_to_base64(url=openai_image_url) - - if openai_image_url.startswith("data:image/"): - # Extract the base64 image data - base64_data = openai_image_url.split("data:image/")[1].split(";base64,")[1] - else: - base64_data = openai_image_url - - return base64_data - except Exception as e: - if "Error: Unable to fetch image from URL" in str(e): - raise e - raise Exception( - """Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{base64_image}". """ - ) - - -def _handle_ollama_system_message(messages: list, prompt: str, msg_i: int) -> tuple[str, int]: - system_content_str = "" - ## MERGE CONSECUTIVE SYSTEM CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - msg_content = convert_content_list_to_str(messages[msg_i]) - system_content_str += msg_content - - msg_i += 1 - - return system_content_str, msg_i - - -def ollama_pt( - model: str, messages: list -) -> ( - str | OllamaVisionModelObject -): # https://github.com/ollama/ollama/blob/af4cf55884ac54b9e637cd71dadfe9b7a5685877/docs/modelfile.md#template - user_message_types: Final = {"user", "tool", "function"} - msg_i = 0 - images: Final = [] - prompt = "" - while msg_i < len(messages): - init_msg_i = msg_i - user_content_str = "" - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - msg_content = messages[msg_i].get("content") - if msg_content: - if isinstance(msg_content, list): - for m in msg_content: - if m.get("type", "") == "image_url": - if isinstance(m["image_url"], str): - images.append(m["image_url"]) - elif isinstance(m["image_url"], dict): - images.append(m["image_url"]["url"]) - elif m.get("type", "") == "text": - user_content_str += m["text"] - else: - # Tool message content will always be a string - user_content_str += msg_content - - msg_i += 1 - - if user_content_str: - prompt += f"### User:\n{user_content_str}\n\n" - - system_content_str, msg_i = _handle_ollama_system_message(messages, prompt, msg_i) - if system_content_str: - prompt += f"### System:\n{system_content_str}\n\n" - - assistant_content_str = "" - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_content_str += convert_content_list_to_str(messages[msg_i]) - - tool_calls = messages[msg_i].get("tool_calls") - ollama_tool_calls = [] - if tool_calls: - for call in tool_calls: - call_id: str = call["id"] - function_name: str = call["function"]["name"] - arguments = json.loads(call["function"]["arguments"]) - - ollama_tool_calls.append( - { - "id": call_id, - "type": "function", - "function": { - "name": function_name, - "arguments": arguments, - }, - } - ) - - if ollama_tool_calls: - assistant_content_str += f"Tool Calls: {json.dumps(ollama_tool_calls, indent=2)}" - - msg_i += 1 - - if assistant_content_str: - prompt += f"### Assistant:\n{assistant_content_str}\n\n" - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider="ollama", - ) - - response_dict: Final[OllamaVisionModelObject] = { - "prompt": prompt, - "images": images, - } - - return response_dict - - -def mistral_instruct_pt(messages): - # Following the Mistral example's https://huggingface.co/docs/transformers/main/chat_templating - prompt: Final = custom_prompt( - initial_prompt_value="", - role_dict={ - "system": { - "pre_message": "[INST] \n", - "post_message": " [/INST]\n", - }, - "user": {"pre_message": "[INST] ", "post_message": " [/INST]\n"}, - "assistant": {"pre_message": " ", "post_message": " "}, - }, - final_prompt_value="", - messages=messages, - ) - return prompt - - -# Falcon prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 -def falcon_instruct_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += message["content"] - else: - prompt += message["role"] + ":" + message["content"].replace("\r\n", "\n").replace("\n\n", "\n") - prompt += "\n\n" - - return prompt - - -def falcon_chat_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "System: " + message["content"] - elif message["role"] == "assistant": - prompt += "Falcon: " + message["content"] - elif message["role"] == "user": - prompt += "User: " + message["content"] - - return prompt - - -# MPT prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 -def mpt_chat_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "<|im_start|>system" + message["content"] + "<|im_end|>" + "\n" - elif message["role"] == "assistant": - prompt += "<|im_start|>assistant" + message["content"] + "<|im_end|>" + "\n" - elif message["role"] == "user": - prompt += "<|im_start|>user" + message["content"] + "<|im_end|>" + "\n" - return prompt - - -# WizardCoder prompt template - https://huggingface.co/WizardLM/WizardCoder-Python-34B-V1.0#prompt-format -def wizardcoder_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += message["content"] + "\n\n" - elif message["role"] == "user": # map to 'Instruction' - prompt += "### Instruction:\n" + message["content"] + "\n\n" - elif message["role"] == "assistant": # map to 'Response' - prompt += "### Response:\n" + message["content"] + "\n\n" - return prompt - - -# Phind-CodeLlama prompt template - https://huggingface.co/Phind/Phind-CodeLlama-34B-v2#how-to-prompt-the-model -def phind_codellama_pt(messages): - prompt = "" - for message in messages: - if message["role"] == "system": - prompt += "### System Prompt\n" + message["content"] + "\n\n" - elif message["role"] == "user": - prompt += "### User Message\n" + message["content"] + "\n\n" - elif message["role"] == "assistant": - prompt += "### Assistant\n" + message["content"] + "\n\n" - return prompt - - -def _render_chat_template(env, chat_template: str, bos_token: str, eos_token: str, messages: list) -> str: - """ - Shared template rendering logic for both sync and async hf_chat_template - - Args: - env: Jinja2 environment - chat_template: Chat template string - bos_token: Beginning of sequence token - eos_token: End of sequence token - messages: Messages to render - - Returns: - Rendered template string - """ - try: - template: Final = env.from_string(chat_template) - except Exception as e: - raise e - - def _is_system_in_template(): - try: - # Try rendering the template with a system message - template.render( - messages=[{"role": "system", "content": "test"}], - eos_token="", - bos_token="", - ) - return True - # This will be raised if Jinja attempts to render the system message and it can't - except Exception: - return False - - try: - rendered_text = "" - # Render the template with the provided values - if _is_system_in_template(): - rendered_text = template.render( - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - add_generation_prompt=True, - ) - else: - # treat a system message as a user message, if system not in template - reformatted_messages: Final = [] - try: - for message in messages: - if message["role"] == "system": - reformatted_messages.append({"role": "user", "content": message["content"]}) - else: - reformatted_messages.append(message) - rendered_text = template.render( - bos_token=bos_token, - eos_token=eos_token, - messages=reformatted_messages, - add_generation_prompt=True, - ) - except Exception as e: - if "Conversation roles must alternate user/assistant" in str(e): - # reformat messages to ensure user/assistant are alternating - new_messages: Final = [] - for i in range(len(reformatted_messages) - 1): - new_messages.append(reformatted_messages[i]) - if reformatted_messages[i]["role"] == reformatted_messages[i + 1]["role"]: - if reformatted_messages[i]["role"] == "user": - new_messages.append({"role": "assistant", "content": ""}) - else: - new_messages.append({"role": "user", "content": ""}) - new_messages.append(reformatted_messages[-1]) - rendered_text = template.render(bos_token=bos_token, eos_token=eos_token, messages=new_messages) - - return rendered_text - except Exception as e: - raise Exception(f"Error rendering template - {e}") # don't use verbose_logger.exception, if exception is raised - - -async def _afetch_and_extract_template( - model: str, chat_template: str | None, get_config_fn, get_template_fn -) -> tuple[str, str, str]: - """ - Async version: Fetch template and tokens from HuggingFace. - - Returns: (chat_template, bos_token, eos_token) - """ - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _extract_token_value, - ) - - bos_token = "" - eos_token = "" - - if chat_template is None: - # Fetch or retrieve cached tokenizer config - if model in litellm.known_tokenizer_config: - tokenizer_config = litellm.known_tokenizer_config[model] - else: - tokenizer_config = await get_config_fn(hf_model_name=model) - litellm.known_tokenizer_config.update({model: tokenizer_config}) - - # Try to get chat template from tokenizer_config.json first - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - and "chat_template" in tokenizer_config["tokenizer"] - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - chat_template = tokenizer_data["chat_template"] - else: - # Fallback: Try to fetch chat template from separate .jinja file - template_result: Final = await get_template_fn(hf_model_name=model) - if template_result.get("status") == "success": - chat_template = template_result["chat_template"] - # Still try to get tokens from tokenizer_config if available - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - else: - raise Exception("No chat template found") - - return chat_template, bos_token, eos_token - - -def _fetch_and_extract_template( - model: str, chat_template: str | None, get_config_fn, get_template_fn -) -> tuple[str, str, str]: - """ - Sync version: Fetch template and tokens from HuggingFace. - - Returns: (chat_template, bos_token, eos_token) - """ - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _extract_token_value, - ) - - bos_token = "" - eos_token = "" - - if chat_template is None: - # Fetch or retrieve cached tokenizer config - if model in litellm.known_tokenizer_config: - tokenizer_config = litellm.known_tokenizer_config[model] - else: - tokenizer_config = get_config_fn(hf_model_name=model) - litellm.known_tokenizer_config.update({model: tokenizer_config}) - - # Try to get chat template from tokenizer_config.json first - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - and "chat_template" in tokenizer_config["tokenizer"] - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - chat_template = tokenizer_data["chat_template"] - else: - # Fallback: Try to fetch chat template from separate .jinja file - template_result: Final = get_template_fn(hf_model_name=model) - if template_result.get("status") == "success": - chat_template = template_result["chat_template"] - # Still try to get tokens from tokenizer_config if available - if ( - tokenizer_config.get("status") == "success" - and "tokenizer" in tokenizer_config - and isinstance(tokenizer_config["tokenizer"], dict) - ): - tokenizer_data: dict = tokenizer_config["tokenizer"] - bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) - eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) - else: - raise Exception("No chat template found") - - return chat_template, bos_token, eos_token - - -async def ahf_chat_template(model: str, messages: list, chat_template: str | None = None): - """HuggingFace chat template (async version)""" - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _aget_chat_template_file, - _aget_tokenizer_config, - strftime_now, - ) - - env: Final = ImmutableSandboxedEnvironment() - env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") - env.globals["strftime_now"] = strftime_now - - template, bos_token, eos_token = await _afetch_and_extract_template( - model=model, - chat_template=chat_template, - get_config_fn=_aget_tokenizer_config, - get_template_fn=_aget_chat_template_file, - ) - return _render_chat_template( - env=env, - chat_template=template, - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - ) - - -def hf_chat_template(model: str, messages: list, chat_template: str | None = None): - """HuggingFace chat template (sync version)""" - from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( - _get_chat_template_file, - _get_tokenizer_config, - strftime_now, - ) - - env: Final = ImmutableSandboxedEnvironment() - env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") - env.globals["strftime_now"] = strftime_now - - template, bos_token, eos_token = _fetch_and_extract_template( - model=model, - chat_template=chat_template, - get_config_fn=_get_tokenizer_config, - get_template_fn=_get_chat_template_file, - ) - return _render_chat_template( - env=env, - chat_template=template, - bos_token=bos_token, - eos_token=eos_token, - messages=messages, - ) - - -def deepseek_r1_pt(messages): - return hf_chat_template(model="deepseek-r1/deepseek-r1-7b-instruct", messages=messages) - - -# Anthropic template -def claude_2_1_pt( - messages: list, -): # format - https://docs.anthropic.com/claude/docs/how-to-use-system-prompts - """ - Claude v2.1 allows system prompts (no Human: needed), but requires it be followed by Human: - - you can't just pass a system message - - you can't pass a system message and follow that with an assistant message - if system message is passed in, you can only do system, human, assistant or system, human - - if a system message is passed in and followed by an assistant message, insert a blank human message between them. - - Additionally, you can "put words in Claude's mouth" by ending with an assistant message. - See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth - """ - - class AnthropicConstants(Enum): - HUMAN_PROMPT = "\n\nHuman: " - AI_PROMPT = "\n\nAssistant: " - - prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "user": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{message['content']}" - elif message["role"] == "assistant": - if idx > 0 and messages[idx - 1]["role"] == "system": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}" # Insert a blank human message - prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" - if messages[-1]["role"] != "assistant": - prompt += f"{AnthropicConstants.AI_PROMPT.value}" # prompt must end with \"\n\nAssistant: " turn - return prompt - - -### IBM Granite - - -def ibm_granite_pt(messages: list): - """ - IBM's Granite models uses the template: - <|system|> {system_message} <|user|> {user_message} <|assistant|> {assistant_message} - - See: https://www.ibm.com/docs/en/watsonx-as-a-service?topic=solutions-supported-foundation-models - """ - return custom_prompt( - messages=messages, - role_dict={ - "system": { - "pre_message": "<|system|>\n", - "post_message": "\n", - }, - "user": { - "pre_message": "<|user|>\n", - # Assistant tag is needed in the prompt after the user message - # to avoid the model completing the users sentence before it answers - # https://www.ibm.com/docs/en/watsonx/w-and-w/2.0.x?topic=models-granite-13b-chat-v2-prompting-tips#chat - "post_message": "\n<|assistant|>\n", - }, - "assistant": { - "pre_message": "", - "post_message": "\n", - }, - }, - ).strip() - - -### ANTHROPIC ### - - -def anthropic_pt( - messages: list, -): # format - https://docs.anthropic.com/claude/reference/complete_post - """ - You can "put words in Claude's mouth" by ending with an assistant message. - See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth - """ - - class AnthropicConstants(Enum): - HUMAN_PROMPT = "\n\nHuman: " - AI_PROMPT = "\n\nAssistant: " - - prompt = "" - for idx, message in enumerate(messages): # needs to start with `\n\nHuman: ` and end with `\n\nAssistant: ` - if message["role"] == "user": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" - else: - prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" - if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` - prompt = f"{AnthropicConstants.HUMAN_PROMPT.value}" + prompt - if messages[-1]["role"] != "assistant": - prompt += f"{AnthropicConstants.AI_PROMPT.value}" - return prompt - - -def construct_format_parameters_prompt(parameters: dict): - parameter_str = "\n" - for k, v in parameters.items(): - parameter_str += f"<{k}>" - parameter_str += f"{v}" - parameter_str += f"" - parameter_str += "\n" - return parameter_str - - -def construct_format_tool_for_claude_prompt(name, description, parameters): - constructed_prompt: Final = ( - "\n" - f"{name}\n" - "\n" - f"{description}\n" - "\n" - "\n" - f"{construct_format_parameters_prompt(parameters)}\n" - "\n" - "" - ) - return constructed_prompt - - -def construct_tool_use_system_prompt( - tools, -): # from https://github.com/anthropics/anthropic-cookbook/blob/main/function_calling/function_calling.ipynb - tool_str_list: Final = [] - for tool in tools: - tool_function = get_attribute_or_key(tool, "function") - tool_str = construct_format_tool_for_claude_prompt( - get_attribute_or_key(tool_function, "name"), - get_attribute_or_key(tool_function, "description", ""), - get_attribute_or_key(tool_function, "parameters", {}), - ) - tool_str_list.append(tool_str) - tool_use_system_prompt: Final = ( - "In this environment you have access to a set of tools you can use to answer the user's question.\n" - "\n" - "You may call them like this:\n" - "\n" - "\n" - "$TOOL_NAME\n" - "\n" - "<$PARAMETER_NAME>$PARAMETER_VALUE\n" - "...\n" - "\n" - "\n" - "\n" - "\n" - "Here are the tools available:\n" - "\n" + "\n".join([tool_str for tool_str in tool_str_list]) + "\n" - ) - return tool_use_system_prompt - - -def convert_generic_image_chunk_to_openai_image_obj( - image_chunk: GenericImageParsingChunk, -) -> str: - """ - Convert a generic image chunk to an OpenAI image object. - - Input: - GenericImageParsingChunk( - type="base64", - media_type="image/jpeg", - data="...", - ) - - Return: - "data:image/jpeg;base64,{base64_image}" - """ - media_type: Final = image_chunk["media_type"] - return "data:{};{},{}".format(media_type, image_chunk["type"], image_chunk["data"]) - - -def convert_to_anthropic_image_obj(openai_image_url: str, format: str | None) -> GenericImageParsingChunk: - """ - Input: - "image_url": "data:image/jpeg;base64,{base64_image}", - - Return: - "source": { - "type": "base64", - "media_type": "image/jpeg", - "data": {base64_image}, - } - """ - try: - if openai_image_url.startswith("http"): - openai_image_url = convert_url_to_base64(url=openai_image_url) - # Extract the media type and base64 data - media_type, base64_data = openai_image_url.split("data:")[1].split(";base64,") - - if format: - media_type = format - else: - media_type = media_type.replace("\\/", "/") - - return GenericImageParsingChunk( - type="base64", - media_type=media_type, - data=base64_data, - ) - except litellm.ImageFetchError: - raise - except Exception as e: - raise Exception( - f"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{{base64_image}}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']. Error: {e}""" - ) - - -def create_anthropic_image_param( - image_url_input: str | dict, - format: str | None = None, - is_bedrock_invoke: bool = False, -) -> AnthropicMessagesImageParam: - """ - Create an AnthropicMessagesImageParam from an image URL input. - - Supports both URL references (for HTTP/HTTPS URLs) and base64 encoding. - """ - # Extract URL and format from input - if isinstance(image_url_input, str): - image_url = image_url_input - else: - image_url = image_url_input.get("url", "") - if format is None: - format = image_url_input.get("format") - - # Check if the image URL is an HTTP/HTTPS URL - if image_url.startswith("http://") or image_url.startswith("https://"): - # For Bedrock invoke and Vertex AI Anthropic, always convert URLs to base64 - # as these providers don't support URL sources for images - if is_bedrock_invoke or image_url.startswith("http://"): - base64_url: Final = convert_url_to_base64(url=image_url) - image_chunk = convert_to_anthropic_image_obj(openai_image_url=base64_url, format=format) - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - else: - # HTTPS URL - pass directly for regular Anthropic - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSourceUrl( - type="url", - url=image_url, - ), - ) - else: - # Convert to base64 for data URIs or other formats - image_chunk = convert_to_anthropic_image_obj(openai_image_url=image_url, format=format) - return AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - - -# The following XML functions will be deprecated once JSON schema support is available on Bedrock and Vertex -# ------------------------------------------------------------------------------ -def convert_to_anthropic_tool_result_xml(message: dict) -> str: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - """ - - """ - Anthropic tool_results look like: - - [Successful results] - - - get_current_weather - - function result goes here - - - - - [Error results] - - - error message goes here - - - """ - name: Final = message.get("name") - content = message.get("content", "") - content = content.replace("<", "<").replace(">", ">").replace("&", "&") - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - anthropic_tool_result: Final = ( - "\n" - "\n" - f"{name}\n" - "\n" - f"{content}\n" - "\n" - "\n" - "" - ) - - return anthropic_tool_result - - -def convert_to_anthropic_tool_invoke_xml(tool_calls: list) -> str: - invokes = "" - for tool in tool_calls: - if get_attribute_or_key(tool, "type") != "function": - continue - - tool_function = get_attribute_or_key(tool, "function") - tool_name = get_attribute_or_key(tool_function, "name") - tool_arguments = get_attribute_or_key(tool_function, "arguments") - parsed_args = parse_tool_call_arguments( - tool_arguments, tool_name=tool_name, context="Anthropic XML tool invoke" - ) - if isinstance(parsed_args, dict): - parameters = "".join(f"<{param}>{val}\n" for param, val in parsed_args.items()) - else: - parameters = f"{parsed_args}\n" - invokes += f"\n{tool_name}\n\n{parameters}\n\n" - - anthropic_tool_invoke: Final = f"\n{invokes}" - - return anthropic_tool_invoke - - -def anthropic_messages_pt_xml(messages: list): - """ - format messages for anthropic - 1. Anthropic supports roles like "user" and "assistant", (here litellm translates system-> assistant) - 2. The first message always needs to be of role "user" - 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) - 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) - 5. System messages are a separate param to the Messages API (used for tool calling) - 6. Ensure we only accept role, content. (message.name is not supported) - """ - # add role=tool support to allow function call result/error submission - user_message_types: Final = {"user", "tool"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final = [] - msg_i = 0 - while msg_i < len(messages): - user_content = [] - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "image_url": - format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None - image_param = create_anthropic_image_param(m["image_url"], format=format) - # Convert to dict format for XML version - source = image_param["source"] - if isinstance(source, dict) and source.get("type") == "url": - # Type narrowing for URL source - url_source = cast(AnthropicContentParamSourceUrl, source) - user_content.append( - { - "type": "image", - "source": { - "type": "url", - "url": url_source["url"], - }, - } - ) - else: - # Type narrowing for base64 source - base64_source = cast(AnthropicContentParamSource, source) - user_content.append( - { - "type": "image", - "source": { - "type": "base64", - "media_type": base64_source["media_type"], - "data": base64_source["data"], - }, - } - ) - elif m.get("type", "") == "text": - user_content.append({"type": "text", "text": m["text"]}) - else: - # Tool message content will always be a string - user_content.append( - { - "type": "text", - "text": ( - convert_to_anthropic_tool_result_xml(messages[msg_i]) - if messages[msg_i]["role"] == "tool" - else messages[msg_i]["content"] - ), - } - ) - - msg_i += 1 - - if user_content: - new_messages.append({"role": "user", "content": user_content}) - - assistant_content = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_text = messages[msg_i].get("content") or "" # either string or none - if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion - assistant_text += convert_to_anthropic_tool_invoke_xml(messages[msg_i]["tool_calls"]) - - assistant_content.append({"type": "text", "text": assistant_text}) - msg_i += 1 - - if assistant_content: - new_messages.append({"role": "assistant", "content": assistant_content}) - - if not new_messages or new_messages[0]["role"] != "user": - if litellm.modify_params: - new_messages.insert(0, {"role": "user", "content": [{"type": "text", "text": "."}]}) - else: - raise Exception( - "Invalid first message. Should always start with 'role'='user' for Anthropic. System prompt is sent separately for Anthropic. set 'litellm.modify_params = True' or 'litellm_settings:modify_params = True' on proxy, to insert a placeholder user message - '.' as the first message, " - ) - - if new_messages[-1]["role"] == "assistant": - for content in new_messages[-1]["content"]: - if isinstance(content, dict) and content["type"] == "text": - content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message - - return new_messages - - -# ------------------------------------------------------------------------------ - - -def _azure_tool_call_invoke_helper( - function_call_params: ChatCompletionToolCallFunctionChunk, -) -> ChatCompletionToolCallFunctionChunk | None: - """ - Azure requires 'arguments' to be a string. - """ - if function_call_params.get("arguments") is None: - function_call_params["arguments"] = "" - return function_call_params - - -def _azure_image_url_helper(content: ChatCompletionImageObject): - if isinstance(content["image_url"], str): - content["image_url"] = {"url": content["image_url"]} - else: - content["image_url"] = cast( - ChatCompletionImageUrlObject, - {k: v for k, v in content["image_url"].items() if k != "format"}, - ) - - -def _azure_file_helper(content: ChatCompletionFileObject) -> None: - content["file"] = cast( - ChatCompletionFileObjectFile, - {k: v for k, v in content.get("file", {}).items() if k != "format"}, - ) - - -def convert_to_azure_openai_messages( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: - for m in messages: - if m["role"] == "assistant": - function_call = m.get("function_call", None) - if function_call is not None: - m["function_call"] = _azure_tool_call_invoke_helper(function_call) - - if m["role"] == "user" and isinstance(m.get("content"), list): - for content in m.get("content", []): - if isinstance(content, dict) and content.get("type") == "image_url": - _azure_image_url_helper(cast(ChatCompletionImageObject, content)) - elif isinstance(content, dict) and content.get("type") == "file": - _azure_file_helper(cast(ChatCompletionFileObject, content)) - return messages - - -# ------------------------------------------------------------------------------ - - -def infer_protocol_value( - value: object, -) -> Literal[ - "string_value", - "number_value", - "bool_value", - "struct_value", - "list_value", - "null_value", - "unknown", -]: - if value is None: - return "null_value" - if isinstance(value, int) or isinstance(value, float): - return "number_value" - if isinstance(value, str): - return "string_value" - if isinstance(value, bool): - return "bool_value" - if isinstance(value, dict): - return "struct_value" - if isinstance(value, list): - return "list_value" - - return "unknown" - - -def _gemini_tool_call_invoke_helper( - function_call_params: ChatCompletionToolCallFunctionChunk, - tool_call_id: str | None = None, -) -> VertexFunctionCall | None: - name: Final = function_call_params.get("name", "") or "" - arguments: Final = function_call_params.get("arguments", "") - if ( - isinstance(arguments, str) and len(arguments) == 0 - ): # pass empty dict, if arguments is empty string - prevents call from failing - arguments_dict = { - "type": "object", - } - else: - arguments_dict = json.loads(arguments) - function_call: Final = VertexFunctionCall( - name=name, - args=arguments_dict, - ) - if tool_call_id: - clean_id: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] - if clean_id: - function_call["id"] = clean_id - return function_call - - -def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: str | None) -> str: - """ - Embed thought signature into tool call ID for OpenAI client compatibility. - - Args: - tool_call_id: The tool call ID (e.g., "call_abc123...") - thought_signature: Base64-encoded signature from Gemini response - - Returns: - Tool call ID with embedded signature if present, otherwise original ID - Format: call___thought__ - - See: https://ai.google.dev/gemini-api/docs/thought-signatures - """ - if thought_signature: - return f"{tool_call_id}{THOUGHT_SIGNATURE_SEPARATOR}{thought_signature}" - return tool_call_id - - -def _get_thought_signature_from_tool(tool: dict) -> str | None: - """Extract thought signature from tool call's provider_specific_fields. - - If not provided try to extract thought signature from tool call id - - Checks both tool.provider_specific_fields and tool.function.provider_specific_fields. - Returns None when the tool call carries no signature; callers decide whether a - placeholder signature is needed. - """ - # First check tool's provider_specific_fields - provider_fields: Final = tool.get("provider_specific_fields") or {} - if isinstance(provider_fields, dict): - signature = provider_fields.get("thought_signature") - if signature: - return signature - - # Then check function's provider_specific_fields - function: Final = tool.get("function") - if function: - if isinstance(function, dict): - func_provider_fields: Final = function.get("provider_specific_fields") or {} - if isinstance(func_provider_fields, dict): - signature = func_provider_fields.get("thought_signature") - if signature: - return signature - elif hasattr(function, "provider_specific_fields") and function.provider_specific_fields: - if isinstance(function.provider_specific_fields, dict): - signature = function.provider_specific_fields.get("thought_signature") - if signature: - return signature - # Check if thought signature is embedded in tool call ID - tool_call_id: Final = tool.get("id") - if tool_call_id and THOUGHT_SIGNATURE_SEPARATOR in tool_call_id: - parts: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1) - if len(parts) == 2: - _, signature = parts - return signature - return None - - -def _get_dummy_thought_signature() -> str: - """Generate a dummy thought signature for models that require it. - - This is used when transferring conversation history from older models - (like gemini-2.5-flash) to gemini-3, which requires thought_signature - for strict validation. Google documents it as a last resort that "will - negatively impact model performance", so callers must only fall back to it - when no real signature is available. - - See: - https://ai.google.dev/gemini-api/docs/thought-signatures#faqs - https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures - """ - dummy_data: Final = b"skip_thought_signature_validator" - return base64.b64encode(dummy_data).decode("utf-8") - - -def convert_to_gemini_tool_call_invoke( - message: ChatCompletionAssistantMessage, - model: str | None = None, - forward_function_call_id: bool = False, -) -> list[VertexPartType]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - """ - Gemini tool call invokes: - { - "role": "model", - "parts": [ - { - "functionCall": { - "name": "get_current_weather", - "args": { - "unit": "fahrenheit", - "predicted_temperature": 45, - "location": "Boston, MA", - } - } - } - ] - } - """ - - """ - - json.load the arguments - """ - try: - _parts_list: Final[list[VertexPartType]] = [] - tool_calls: Final = message.get("tool_calls", None) - function_call: Final = message.get("function_call", None) - - from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( - VertexGeminiConfig, - ) - - needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model) - - if tool_calls is not None: - for tool in tool_calls: - if "function" in tool: - gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper( - function_call_params=tool["function"], - tool_call_id=(tool.get("id") if forward_function_call_id else None), - ) - if gemini_function_call is not None: - part_dict: VertexPartType = {"function_call": gemini_function_call} - thought_signature = _get_thought_signature_from_tool(dict(tool)) - # Gemini signs only the first functionCall part of a parallel batch, so scope the - # placeholder fallback to that part instead of fabricating one per sibling call: - # https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example - is_first_function_call = len(_parts_list) == 0 - if not thought_signature and is_first_function_call and needs_dummy_signature: - thought_signature = _get_dummy_thought_signature() - if thought_signature: - part_dict["thoughtSignature"] = thought_signature - - _parts_list.append(part_dict) - else: # don't silently drop params. Make it clear to user what's happening. - raise Exception( - f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {tool}" - ) - elif function_call is not None: - gemini_function_call = _gemini_tool_call_invoke_helper(function_call_params=function_call) - if gemini_function_call is not None: - part_dict_function: Final[VertexPartType] = {"function_call": gemini_function_call} - - # Extract thought signature from function_call's provider_specific_fields - thought_signature = None - provider_fields: Final = ( - function_call.get("provider_specific_fields") if isinstance(function_call, dict) else {} - ) - if isinstance(provider_fields, dict): - thought_signature = provider_fields.get("thought_signature") - - # If no signature found and model is gemini-3, use dummy signature - if not thought_signature and needs_dummy_signature: - thought_signature = _get_dummy_thought_signature() - - if thought_signature: - part_dict_function["thoughtSignature"] = thought_signature - - _parts_list.append(part_dict_function) - else: # don't silently drop params. Make it clear to user what's happening. - raise Exception( - f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {message}" - ) - return _parts_list - except Exception as e: - raise Exception(f"Unable to convert openai tool calls={message} to gemini tool calls. Received error={e}") - - -def convert_to_gemini_tool_call_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - last_message_with_tool_calls: dict | None, - forward_function_call_id: bool = False, -) -> VertexPartType | list[VertexPartType]: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "content": "function result goes here", - }, - - # NOTE: Function messages have been deprecated - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - - Supports content with images for Computer Use: - { - "role": "tool", - "tool_call_id": "call_abc123", - "content": [ - {"type": "text", "text": "I found the requested image:"}, - {"type": "input_image", "image_url": "https://example.com/image.jpg" } - ] - } - """ - from litellm.types.llms.vertex_ai import BlobType - - content_str: str = "" - inline_data_list: Final[list[BlobType]] = [] - - if "content" in message: - if isinstance(message["content"], str): - content_str = message["content"] - # Detect data-URL images (e.g. from Anthropic tool_result with a single image block - # that was serialised as a plain string by translate_anthropic_messages_to_openai) - # and promote them to inline_data so Gemini receives actual image bytes. - if content_str[:5].lower() == "data:" and ";base64," in content_str: - try: - mime_rest: Final = content_str[5:].split(";base64,", 1) - if len(mime_rest) == 2 and mime_rest[0].startswith("image/"): - # Strip any extra parameters (e.g. ";charset=UTF-8") from the MIME segment - clean_mime: Final = mime_rest[0].split(";")[0].strip() - inline_data_list.append(BlobType(data=mime_rest[1], mime_type=clean_mime)) - content_str = "" - except Exception as e: - verbose_logger.warning("Failed to parse data URL in tool response: %s", e) - elif isinstance(message["content"], list): - content_list: Final = message["content"] - for content in content_list: - content_type = content.get("type", "") - if content_type == "text": - content_str += content.get("text", "") - elif content_type == "image": # pyright: ignore[reportUnnecessaryComparison] # loose runtime dict - # Anthropic-native image block: {"type": "image", "source": {"type": "base64", ...}} - source = content.get("source", {}) - if isinstance(source, dict) and source.get("type") == "base64": - try: - inline_data_list.append( - BlobType( - data=source.get("data", ""), - mime_type=source.get("media_type", "image/jpeg"), - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process Anthropic image block in tool response: %s", e) - elif content_type in ("input_image", "image_url"): - # Extract image for inline_data (for Computer Use screenshots and tool results) - image_url_data = content.get("image_url", "") - image_url = image_url_data.get("url", "") if isinstance(image_url_data, dict) else image_url_data - - if image_url: - # Convert image to base64 blob format for Gemini - try: - image_obj = convert_to_anthropic_image_obj(image_url, format=None) - inline_data_list.append( - BlobType( - data=image_obj["data"], - mime_type=image_obj["media_type"], - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process image in tool response: %s", e) - elif content_type in ("file", "input_file"): # pyright: ignore[reportUnnecessaryContains] # loose runtime dict - # Extract file for inline_data (for tool results with PDF, audio, video, etc.) - file_data = content.get("file_data", "") - if not file_data: - file_content = content.get("file", {}) - file_data = ( - file_content.get("file_data", "") - if isinstance(file_content, dict) - else file_content - if isinstance(file_content, str) - else "" - ) - - if file_data: - # Convert file to base64 blob format for Gemini - try: - file_obj = convert_to_anthropic_image_obj(file_data, format=None) - inline_data_list.append( - BlobType( - data=file_obj["data"], - mime_type=file_obj["media_type"], - ) - ) - except Exception as e: - verbose_logger.warning("Failed to process file in tool response: %s", e) - name: str | None = message.get("name", "") - - # Recover name from last message with tool calls - if last_message_with_tool_calls: - tools: Final = last_message_with_tool_calls.get("tool_calls", []) - msg_tool_call_id: Final = message.get("tool_call_id", None) - for tool in tools: - prev_tool_call_id = tool.get("id", None) - if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: - name = tool.get("function", {}).get("name", "") - - # Echo the OpenAI tool_call_id on functionResponse (strip thought-signature suffix). - gemini_call_id: str | None = None - if forward_function_call_id: - raw_tool_call_id: Final = message.get("tool_call_id") - if raw_tool_call_id and isinstance(raw_tool_call_id, str): - stripped_id: Final = raw_tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] - if stripped_id: - gemini_call_id = stripped_id - - if not name: - raise Exception( - f"Missing corresponding tool call for tool response message. Received - message={message}, last_message_with_tool_calls={last_message_with_tool_calls}" - ) - - # Parse response data - support both JSON string and plain string - # For Computer Use, the response should contain structured data like {"url": "..."} - response_data: dict - try: - if content_str.strip().startswith("{") or content_str.strip().startswith("["): - # Try to parse as JSON (for Computer Use structured responses) - parsed: Final = json.loads(content_str) - if isinstance(parsed, dict): - response_data = parsed # Use the parsed JSON directly - else: - response_data = {"content": content_str} - else: - response_data = {"content": content_str} - except (json.JSONDecodeError, ValueError): - # Not valid JSON, wrap in content field - response_data = {"content": content_str} - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - _function_response: Final = VertexFunctionResponse( - name=name, - response=response_data, - ) - if gemini_call_id: - _function_response["id"] = gemini_call_id - - _part: Final[VertexPartType] = {"function_response": _function_response} - - # For multimodal function responses, Gemini expects media parts nested - # inside functionResponse.parts instead of sibling content parts. - if inline_data_list: - _function_response["parts"] = [{"inline_data": inline_data} for inline_data in inline_data_list] - return [_part] - - return _part - - -_TOOL_USE_ID_FALLBACK: Final = "tool_use_id" -_ANTHROPIC_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_-]") -_BEDROCK_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_.:-]") -_BEDROCK_TOOL_USE_ID_MAX_LEN: Final = 64 -_BEDROCK_TOOL_USE_ID_HASH_LEN: Final = 8 - - -def _replace_invalid_tool_use_id_chars(tool_use_id: str, invalid_chars: re.Pattern[str]) -> str: - return invalid_chars.sub("_", tool_use_id) or _TOOL_USE_ID_FALLBACK - - -def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str: - """Anthropic requires tool_use_id to match ^[a-zA-Z0-9_-]+$.""" - return _replace_invalid_tool_use_id_chars(tool_use_id, _ANTHROPIC_TOOL_USE_ID_INVALID_CHARS) - - -def _sanitize_bedrock_tool_use_id(tool_use_id: str) -> str: - """ - Bedrock Converse requires toolUseId to match [a-zA-Z0-9_.:-]+ and be at most 64 chars. - Ids that need rewriting get a short hash of the original appended so two ids that only - differ in a replaced char or past the cut still map to distinct values. - """ - sanitized: Final = _replace_invalid_tool_use_id_chars(tool_use_id, _BEDROCK_TOOL_USE_ID_INVALID_CHARS) - if sanitized == tool_use_id and len(sanitized) <= _BEDROCK_TOOL_USE_ID_MAX_LEN: - return sanitized - digest: Final = hashlib.sha256(tool_use_id.encode()).hexdigest()[:_BEDROCK_TOOL_USE_ID_HASH_LEN] - return f"{sanitized[: _BEDROCK_TOOL_USE_ID_MAX_LEN - _BEDROCK_TOOL_USE_ID_HASH_LEN - 1]}_{digest}" - - -_ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES: Final = {"application/pdf", "text/plain"} - - -def _is_anthropic_document_data_uri(url: str) -> bool: - # Anthropic's base64 document source accepts only application/pdf and - # text/plain (see select_anthropic_content_block_type_for_file). Routing - # other mimes here would produce a document block the API rejects, so we - # leave them on the image code path. - match: Final = re.match(r"data:([^;,]+)", url) - if not match: - return False - return match.group(1) in _ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES - - -def convert_to_anthropic_tool_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - force_base64: bool = False, -) -> AnthropicMessagesToolResultParam: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - - """ - Anthropic tool_results look like: - { - "role": "user", - "content": [ - { - "type": "tool_result", - "tool_use_id": "toolu_01A09q90qw90lq917835lq9", - "content": "ConnectionError: the weather service API is not available (HTTP 500)", - # "is_error": true - } - ] - } - """ - anthropic_content: ( - str - | list[ - AnthropicMessagesToolResultContent - | AnthropicMessagesImageParam - | AnthropicMessagesDocumentParam - | ToolReference - ] - ) = "" - if isinstance(message["content"], str): - anthropic_content = message["content"] - elif isinstance(message["content"], list): - content_list: Final = message["content"] - anthropic_content_list: list[ - AnthropicMessagesToolResultContent - | AnthropicMessagesImageParam - | AnthropicMessagesDocumentParam - | ToolReference - ] = [] - for content in content_list: - if content["type"] == "text": - # Only include cache_control if explicitly set and not None - # to avoid sending "cache_control": null which breaks some API channels - text_content: AnthropicMessagesToolResultContent = { - "type": "text", - "text": content["text"], - } - cache_control_value = content.get("cache_control") - if cache_control_value is not None: - text_content["cache_control"] = cache_control_value - anthropic_content_list.append(text_content) - elif content["type"] == "image_url": - image_url_value = content["image_url"] - format = image_url_value.get("format") if isinstance(image_url_value, dict) else None - url_str = image_url_value.get("url") if isinstance(image_url_value, dict) else image_url_value - # Data URIs with non-image mime types (e.g. application/pdf) must - # translate to Anthropic document blocks, not image blocks — - # wrapping a PDF in `type: "image"` is rejected by the API. - if isinstance(url_str, str) and _is_anthropic_document_data_uri(url_str): - synth_file_message: ChatCompletionFileObject = { - "type": "file", - "file": {"file_data": url_str}, - } - _document_block = anthropic_process_openai_file_message(synth_file_message) - _document_block = add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, _document_block), - original_content_element=content, - ) - anthropic_content_list.append(cast(AnthropicMessagesDocumentParam, _document_block)) - else: - _anthropic_image_param = create_anthropic_image_param( - image_url_value, - format=format, - is_bedrock_invoke=force_base64, - ) - _anthropic_image_param = add_cache_control_to_content( - anthropic_content_element=_anthropic_image_param, - original_content_element=content, - ) - anthropic_content_list.append(cast(AnthropicMessagesImageParam, _anthropic_image_param)) - elif content["type"] == "tool_reference": - anthropic_content_list.append(ToolReference(type="tool_reference", tool_name=content["tool_name"])) - elif content["type"] == "file": - file_content = cast(ChatCompletionFileObject, content) - _file_block = anthropic_process_openai_file_message(file_content) - _file_block = add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, _file_block), - original_content_element=content, - ) - anthropic_content_list.append(_file_block) - - anthropic_content = anthropic_content_list - anthropic_tool_result: AnthropicMessagesToolResultParam | None = None - ## PROMPT CACHING CHECK ## - cache_control: Final = message.get("cache_control", None) - if message["role"] == "tool": - tool_message: Final[ChatCompletionToolMessage] = message - tool_call_id: str = tool_message["tool_call_id"] - # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ - sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) - - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", - tool_use_id=sanitized_tool_use_id, - content=anthropic_content, - ) - - if message["role"] == "function": - function_message: Final[ChatCompletionFunctionMessage] = message - tool_call_id = function_message.get("tool_call_id") or str(uuid.uuid4()) - # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ - sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) - anthropic_tool_result = AnthropicMessagesToolResultParam( - type="tool_result", - tool_use_id=sanitized_tool_use_id, - content=anthropic_content, - ) - - if anthropic_tool_result is None: - raise Exception(f"Unable to parse anthropic tool result for message: {message}") - if cache_control is not None: - anthropic_tool_result["cache_control"] = cache_control - return anthropic_tool_result - - -def convert_function_to_anthropic_tool_invoke( - function_call: dict | ChatCompletionToolCallFunctionChunk, -) -> list[AnthropicMessagesToolUseParam]: - try: - _name: Final = get_attribute_or_key(function_call, "name") or "" - _arguments: Final = get_attribute_or_key(function_call, "arguments") - - tool_input: Final = parse_tool_call_arguments( - _arguments, tool_name=_name, context="Anthropic function to tool invoke" - ) - - anthropic_tool_invoke: Final = [ - AnthropicMessagesToolUseParam( - type="tool_use", - id=str(uuid.uuid4()), - name=_name, - input=tool_input, - ) - ] - return anthropic_tool_invoke - except Exception as e: - raise e - - -def _find_server_tool_result( - tool_id: str, - web_search_results: Sequence[object] | None, - tool_results: Sequence[object] | None, -) -> dict[str, object] | None: - candidates: Final = (*(web_search_results or ()), *(tool_results or ())) - return next( - (result for result in candidates if isinstance(result, dict) and result.get("tool_use_id") == tool_id), - None, - ) - - -def convert_to_anthropic_tool_invoke( - tool_calls: list[ChatCompletionAssistantToolCall], - web_search_results: Sequence[object] | None = None, - tool_results: Sequence[object] | None = None, -) -> list[AnthropicMessagesToolUseParam | dict[str, Any]]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - - """ - Anthropic tool invokes: - { - "role": "assistant", - "content": [ - { - "type": "text", - "text": "To answer this question, I will: 1. Use the get_weather tool to get the current weather in San Francisco. 2. Use the get_time tool to get the current time in the America/Los_Angeles timezone, which covers San Francisco, CA." - }, - { - "type": "tool_use", - "id": "toolu_01A09q90qw90lq917835lq9", - "name": "get_weather", - "input": {"location": "San Francisco, CA"} - } - ] - } - - For server-side tools (web_search), we need to reconstruct: - - server_tool_use blocks (id starts with "srvtoolu_") - - web_search_tool_result blocks (from provider_specific_fields) - - Fixes: https://github.com/BerriAI/litellm/issues/17737 - """ - anthropic_tool_invoke: Final[list[AnthropicMessagesToolUseParam | dict[str, object]]] = [] - - for tool in tool_calls: - if get_attribute_or_key(tool, "type") != "function": - continue - - tool_id = cast(str, get_attribute_or_key(tool, "id")) - tool_name = cast( - str, - get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), - ) - tool_input = parse_tool_call_arguments( - get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments"), - tool_name=tool_name, - context="Anthropic tool invoke", - ) - - server_tool_result = ( - _find_server_tool_result(tool_id, web_search_results, tool_results) - if tool_id.startswith("srvtoolu_") - else None - ) - if server_tool_result is not None: - anthropic_tool_invoke.append( - { - "type": "server_tool_use", - "id": tool_id, - "name": tool_name, - "input": tool_input, - } - ) - anthropic_tool_invoke.append(server_tool_result) - else: - sanitized_tool_id = _sanitize_anthropic_tool_use_id(tool_id) - _anthropic_tool_use_param = AnthropicMessagesToolUseParam( - type="tool_use", - id=sanitized_tool_id, - name=tool_name, - input=tool_input, - ) - - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_tool_use_param, - original_content_element=dict(tool), - ) - - if "cache_control" in _content_element: - _anthropic_tool_use_param["cache_control"] = _content_element["cache_control"] - - anthropic_tool_invoke.append(_anthropic_tool_use_param) - - return anthropic_tool_invoke - - -def add_cache_control_to_content( - anthropic_content_element: dict - | AnthropicMessagesImageParam - | AnthropicMessagesTextParam - | AnthropicMessagesDocumentParam - | AnthropicMessagesToolUseParam - | ChatCompletionThinkingBlock, - original_content_element: dict | AllMessageValues, -): - cache_control_param: Final = original_content_element.get("cache_control") - if cache_control_param is not None and isinstance(cache_control_param, dict): - transformed_param: Final = ChatCompletionCachedContent(**cache_control_param) - - anthropic_content_element["cache_control"] = transformed_param - - return anthropic_content_element - - -def _anthropic_content_element_factory( - image_chunk: GenericImageParsingChunk, -) -> AnthropicMessagesImageParam | AnthropicMessagesDocumentParam: - if image_chunk["media_type"] == "application/pdf": - _anthropic_content_element: AnthropicMessagesDocumentParam | AnthropicMessagesImageParam = ( - AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - ) - else: - _anthropic_content_element = AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - - return _anthropic_content_element - - -def select_anthropic_content_block_type_for_file( - format: str, -) -> Literal["document", "image", "container_upload"]: - if format == "application/pdf" or format == "text/plain": - return "document" - elif format in ["image/jpeg", "image/png", "image/gif", "image/webp"]: - return "image" - else: - return "container_upload" - - -def anthropic_infer_file_id_content_type( - file_id: str, -) -> Literal["document_url", "container_upload"]: - """ - Use when 'format' not provided. - - - URL's - assume are document_url - - Else - assume is container_upload - """ - if file_id.startswith("http") or file_id.startswith("https"): - return "document_url" - else: - return "container_upload" - - -def anthropic_process_openai_file_message( - message: ChatCompletionFileObject, -) -> AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam: - file_message: Final = cast(ChatCompletionFileObject, message) - file_sub: Final = file_message.get("file") - if file_sub is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="anthropic", - ) - file_data: Final = file_sub.get("file_data") - file_id: Final = file_sub.get("file_id") - format: Final = file_sub.get("format") - if file_data: - image_chunk: Final = convert_to_anthropic_image_obj( - openai_image_url=file_data, - format=format, - ) - anthropic_document_param: Final = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSource( - type="base64", - media_type=image_chunk["media_type"], - data=image_chunk["data"], - ), - ) - return anthropic_document_param - elif file_id: - content_block_type: Final = ( - select_anthropic_content_block_type_for_file(format) - if format - else anthropic_infer_file_id_content_type(file_id) - ) - return_block_param: ( - AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam | None - ) = None - if content_block_type == "document": - return_block_param = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSourceFileId( - type="file", - file_id=file_id, - ), - ) - elif content_block_type == "document_url": - return_block_param = AnthropicMessagesDocumentParam( - type="document", - source=AnthropicContentParamSourceUrl( - type="url", - url=file_id, - ), - ) - elif content_block_type == "image": - return_block_param = AnthropicMessagesImageParam( - type="image", - source=AnthropicContentParamSourceFileId( - type="file", - file_id=file_id, - ), - ) - elif content_block_type == "container_upload": - return_block_param = AnthropicMessagesContainerUploadParam(type="container_upload", file_id=file_id) - - if return_block_param is None: - raise Exception(f"Unable to parse anthropic file message: {message}") - return return_block_param - raise Exception(f"Either file_data or file_id must be present in the file message: {message}") - - -_EMPTY_TEXT_PLACEHOLDER: Final = "[System: Empty message content sanitised to satisfy protocol]" - - -def _sanitize_empty_text_content( - message: AllMessageValues, -) -> AllMessageValues: - """ - Case C: Sanitize empty text content - - Replace empty or whitespace-only text content with a placeholder message. - - Handles both string content and list-of-blocks content (rewriting only - the empty text blocks in place; non-text blocks like images are left - untouched). - - Returns: - The message with sanitized content if needed, otherwise the original message - """ - if message.get("role") not in ["user", "assistant"]: - return message - - content: Final = message.get("content") - - if isinstance(content, str): - if not content or not content.strip(): - message = cast(AllMessageValues, dict(message)) # Make a copy - message["content"] = _EMPTY_TEXT_PLACEHOLDER - verbose_logger.debug( - "_sanitize_empty_text_content: Replaced empty text content in %s message", message.get("role") - ) - return message - - if isinstance(content, list): - # Walk the blocks and rewrite any empty text blocks. We rewrite (rather - # than drop) so callers don't end up with an entirely empty content - # list, which Anthropic also rejects. - new_blocks: Final[list[Any]] = [] - rewrote_any = False - for block in content: - if isinstance(block, dict) and block.get("type") == "text": - text = block.get("text") - if not isinstance(text, str) or not text or not text.strip(): - new_block = dict(block) - new_block["text"] = _EMPTY_TEXT_PLACEHOLDER - new_blocks.append(new_block) - rewrote_any = True - continue - new_blocks.append(block) - - if rewrote_any: - message = cast(AllMessageValues, dict(message)) # Make a copy - message["content"] = new_blocks - verbose_logger.debug( - "_sanitize_empty_text_content: Replaced empty text block(s) in %s message", message.get("role") - ) - - return message - - -def _add_missing_tool_results( - current_message: AllMessageValues, - messages: list[AllMessageValues], - current_index: int, -) -> tuple[list[AllMessageValues], int]: - """ - Case A: Missing tool_result for tool_use (orphaned tool calls) - - If an assistant message has tool_calls but no corresponding tool result follows, - add a dummy tool result message indicating the user did not provide the result. - - Returns: - A tuple of: - - List containing the assistant message, followed by existing tool results, - followed by any dummy tool results needed - - Number of original messages consumed (to adjust iteration index) - """ - result_messages: Final[list[AllMessageValues]] = [] - tool_calls: Final = current_message.get("tool_calls") - - if not tool_calls or len(cast(list, tool_calls)) == 0: - return ([current_message], 0) - - # Collect all tool_call_ids from this assistant message - expected_tool_call_ids: Final = set() - for tool_call in cast(list, tool_calls): - tool_call_id = None - if isinstance(tool_call, dict): - tool_call_id = tool_call.get("id") - else: - tool_call_id = getattr(tool_call, "id", None) - if tool_call_id: - expected_tool_call_ids.add(tool_call_id) - - # Collect actual tool result messages that follow this assistant message - found_tool_call_ids: Final = set() - actual_tool_results: Final[list[AllMessageValues]] = [] - j = current_index + 1 - - while j < len(messages): - next_msg = messages[j] - next_role = next_msg.get("role") - - if next_role == "assistant": - break - - if next_role in ["tool", "function"]: - tool_call_id = next_msg.get("tool_call_id") - if tool_call_id and tool_call_id in expected_tool_call_ids: - found_tool_call_ids.add(tool_call_id) - actual_tool_results.append(next_msg) - - j += 1 - - # Find missing tool results - missing_tool_call_ids: Final = expected_tool_call_ids - found_tool_call_ids - - if missing_tool_call_ids: - verbose_logger.debug( - "_add_missing_tool_results: Found %s orphaned tool calls. Adding dummy tool results.", - len(missing_tool_call_ids), - ) - - result_messages.append(current_message) - - # Add existing tool results FIRST - result_messages.extend(actual_tool_results) - - # Then add dummy tool results for missing ones - for tool_call_id in missing_tool_call_ids: - tool_name = "unknown_tool" - for tool_call in cast(list, tool_calls): - tc_id = None - if isinstance(tool_call, dict): - tc_id = tool_call.get("id") - else: - tc_id = getattr(tool_call, "id", None) - - if tc_id == tool_call_id: - if isinstance(tool_call, dict): - function = tool_call.get("function", {}) - if isinstance(function, dict): - tool_name = function.get("name", "unknown_tool") - else: - tool_name = getattr(function, "name", "unknown_tool") - else: - function = getattr(tool_call, "function", None) - if function: - tool_name = getattr(function, "name", "unknown_tool") - break - - dummy_tool_result: ChatCompletionToolMessage = { - "role": "tool", - "tool_call_id": tool_call_id, - "content": f"[System: Tool execution skipped/interrupted by user. No result provided for tool '{tool_name}'.]", - } - result_messages.append(dummy_tool_result) - - # Return the messages and the number of original messages to skip - return (result_messages, len(actual_tool_results)) - - return ([current_message], 0) - - -def _is_orphaned_tool_result( - current_message: AllMessageValues, - sanitized_messages: list[AllMessageValues], -) -> bool: - """ - Case B: Orphaned tool_result (unexpected result) - - Check if a tool message references a tool_call_id that doesn't exist in the previous - assistant message. - - Returns: - True if this is an orphaned tool result that should be removed, False otherwise - """ - if current_message.get("role") not in ["tool", "function"]: - return False - - tool_call_id: Final = current_message.get("tool_call_id") - - if not tool_call_id: - return False - - # Look back to find the most recent assistant message with tool_calls - found_matching_tool_call = False - - for j in range(len(sanitized_messages) - 1, -1, -1): - prev_msg = sanitized_messages[j] - if prev_msg.get("role") == "assistant": - tool_calls = prev_msg.get("tool_calls") - if tool_calls: - for tool_call in cast(list, tool_calls): - tc_id = None - if isinstance(tool_call, dict): - tc_id = tool_call.get("id") - else: - tc_id = getattr(tool_call, "id", None) - - if tc_id == tool_call_id: - found_matching_tool_call = True - break - - break - - if not found_matching_tool_call: - verbose_logger.debug("_is_orphaned_tool_result: Found orphaned tool result with redacted tool_call_id") - return True - - return False - - -def _declared_tool_call_ids(message: Mapping[str, object]) -> frozenset[str]: - tool_calls: Final = message.get("tool_calls") - if not isinstance(tool_calls, list): - return frozenset() - return frozenset( - str(tool_call["id"]) for tool_call in tool_calls if isinstance(tool_call, Mapping) and tool_call.get("id") - ) - - -def group_tool_exchanges(messages: Sequence[Mapping[str, object]]) -> tuple[tuple[int, ...], ...]: - """Group message indices into tool exchanges: an assistant row that made - tool calls, together with the tool rows answering the ids it declared. - - Membership is by ``tool_call_id`` ownership rather than adjacency, so a tool - row belonging to some other call opens its own group instead of being swept - into the exchange it happens to sit next to. Every other row is its own - group. Groups stay contiguous and in order, so a caller can convert or - protect them without reordering the conversation. - - Callers need this because an assistant row and the tool rows answering it - are only well-formed together: ``sanitize_messages_for_tool_calling`` reads - an assistant row whose results are missing as an orphaned tool call, and - a tool row whose call is missing as an orphaned result. - """ - return tuple(_iter_tool_exchange_groups(messages)) - - -def _iter_tool_exchange_groups(messages: Sequence[Mapping[str, object]]) -> Iterator[tuple[int, ...]]: - index = 0 - while index < len(messages): - declared = _declared_tool_call_ids(messages[index]) - end = index + 1 - while ( - declared - and end < len(messages) - and messages[end].get("role") in ("tool", "function") - and str(messages[end].get("tool_call_id")) in declared - ): - end += 1 - yield tuple(range(index, end)) - index = end - - -def sanitize_messages_for_tool_calling( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: - """ - Sanitize messages for tool calling to handle common issues when modify_params=True: - - Case A: Missing tool_result for tool_use (orphaned tool calls) - - If an assistant message has tool_calls but no corresponding tool result follows, - add a dummy tool result message indicating the user did not provide the result. - - Case B: Orphaned tool_result (unexpected result) - - If a tool message references a tool_call_id that doesn't exist in the previous - assistant message, remove that tool message. - - Case C: Empty text content - - Replace empty or whitespace-only text content with a placeholder message. - - Case D: Duplicate tool_result for same tool_use (duplicate results) - - If multiple tool messages reference the same tool_call_id, keep only the last - occurrence. Anthropic requires exactly one tool_result per tool_use and rejects - with: "each tool_use must have a single result". - - This function operates on OpenAI format messages before they are converted to - provider-specific formats. - """ - if not litellm.modify_params: - return messages - - sanitized_messages: list[AllMessageValues] = [] - i = 0 - - while i < len(messages): - current_message = messages[i] - - # Case C: Sanitize empty text content - current_message = _sanitize_empty_text_content(current_message) - - # Case A: Check if assistant message has tool_calls without following tool results - if current_message.get("role") == "assistant": - result_messages, messages_consumed = _add_missing_tool_results(current_message, messages, i) - - # If dummy tool results were added, extend sanitized_messages and skip consumed messages - if len(result_messages) > 1: - sanitized_messages.extend(result_messages) - # Skip the assistant message and any actual tool results that were included - i += 1 + messages_consumed - continue - - # Case B: Check for orphaned tool results - if _is_orphaned_tool_result(current_message, sanitized_messages): - i += 1 - continue # Skip this orphaned tool result - - # Add the message to sanitized list - sanitized_messages.append(current_message) - i += 1 - - # Case D: Deduplicate tool results with the same tool_call_id. - # Anthropic requires exactly one tool_result per tool_use. Session history - # (e.g. from conversation resume) can contain duplicate tool_result messages - # for the same tool_call_id. Keep only the last occurrence *within each - # contiguous block of tool results following an assistant message*. This - # avoids dropping results from earlier turns if a tool_call_id is reused. - # - # NOTE: This intentionally keeps the *last* occurrence (most complete for - # session-resume duplicates), unlike _deduplicate_bedrock_content_blocks - # which keeps the *first*. The Bedrock case handles provider-side content - # block duplication where the first is authoritative; here the duplicate - # arises from history replay where the last entry is the final state. - duplicates_to_remove: Final[set[int]] = set() - seen_in_block: dict[str, int] = {} # tool_call_id -> index (reset per block) - for idx, msg in enumerate(sanitized_messages): - role = msg.get("role") - tcid = msg.get("tool_call_id") if role in ["tool", "function"] else None - if tcid and isinstance(tcid, str): - if tcid in seen_in_block: - # Mark the earlier occurrence for removal (keep latest) - duplicates_to_remove.add(seen_in_block[tcid]) - verbose_logger.warning( - "sanitize_messages_for_tool_calling: dropping duplicate " - "tool_result with tool_call_id=%s. This may indicate " - "duplicate tool messages in conversation history.", - tcid, - ) - seen_in_block[tcid] = idx - elif role not in ("tool", "function"): - # Non-tool message (user, assistant, system) marks a - # conversational-turn boundary — reset tracking. - # Tool/function messages with no tool_call_id are malformed; - # they should NOT reset the block because they don't represent - # a turn boundary and would mask real within-block duplicates. - seen_in_block = {} - - if duplicates_to_remove: - sanitized_messages = [msg for idx, msg in enumerate(sanitized_messages) if idx not in duplicates_to_remove] - - return sanitized_messages - - -def _drop_unsignable_thinking_blocks( - thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], -) -> list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock]: - return [block for block in thinking_blocks if not is_unsignable_thinking_block(block)] - - -_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues] - - -def anthropic_messages_pt( - messages: list[AllMessageValues], - model: str, - llm_provider: str, -) -> _AnthropicMessageList: - """ - format messages for anthropic - 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately). - Models flagged ``supports_mid_conversation_system`` also accept "system" inside - messages after a user turn; the caller decides placement, this keeps such messages. - 2. The first message always needs to be of role "user" - 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) - 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) - 5. System messages are a separate param to the Messages API - 6. Ensure we only accept role, content. (message.name is not supported) - """ - # Sanitize messages for tool calling issues when modify_params=True - messages = sanitize_messages_for_tool_calling(messages) - - # Anthropic rejects empty text content blocks with: - # "messages: text content blocks must be non-empty" - # OpenAI/other providers silently tolerate `{"role": "user", "content": ""}`, - # so callers (and upstream agent frameworks like pydantic-ai) routinely - # send empty user/assistant turns. We always rewrite these to a placeholder - # for Anthropic-shaped requests, independent of `litellm.modify_params`, - # because there is no way to "pass through" an empty text block — the - # request will always 400 otherwise. The richer tool-call sanitization - # (Cases A/B/D in `sanitize_messages_for_tool_calling`) remains gated on - # `modify_params` because it actually mutates conversation structure. - messages = [_sanitize_empty_text_content(m) for m in messages] - - # add role=tool support to allow function call result/error submission - user_message_types: Final = {"user", "tool", "function"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract - - if len(messages) == 0: - if not litellm.modify_params: - raise litellm.BadRequestError( - message=f"Anthropic requires at least one non-system message. Either provide one, or set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add the dummy user message - {DEFAULT_USER_CONTINUE_MESSAGE_TYPED}.", - model=model, - llm_provider=llm_provider, - ) - else: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE_TYPED) - - # Bedrock invoke models have format: invoke/... - # Vertex AI Anthropic also doesn't support URL sources for images - is_bedrock_invoke = model.lower().startswith("invoke/") - is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False - force_base64 = is_bedrock_invoke or is_vertex_ai - - msg_i = 0 - while msg_i < len(messages): - user_content: list[AnthropicMessagesUserMessageValues] = [] - init_msg_i = msg_i - if isinstance(messages[msg_i], BaseModel): - messages[msg_i] = dict(messages[msg_i]) - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - user_message_types_block: ( - ChatCompletionToolMessage | ChatCompletionUserMessage | ChatCompletionFunctionMessage - ) = messages[msg_i] - if user_message_types_block["role"] == "user": - if isinstance(user_message_types_block["content"], list): - user_content_start: int = len(user_content) - for m in user_message_types_block["content"]: - if m.get("type", "") == "image_url": - m = cast(ChatCompletionImageObject, m) - format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None - # Convert ChatCompletionImageUrlObject to dict if needed - image_url_value = m["image_url"] - if isinstance(image_url_value, str): - image_url_input: str | dict[str, object] = image_url_value - else: - # ChatCompletionImageUrlObject or dict case - convert to dict - image_url_input = { - "url": image_url_value["url"], - "format": image_url_value.get("format"), - } - # Bedrock invoke models have format: invoke/... - # Vertex AI Anthropic also doesn't support URL sources for images - is_bedrock_invoke = model.lower().startswith("invoke/") - is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False - force_base64 = is_bedrock_invoke or is_vertex_ai - _anthropic_content_element = create_anthropic_image_param( - image_url_input, - format=format, - is_bedrock_invoke=force_base64, - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_content_element, - original_content_element=dict(m), - ) - - if "cache_control" in _content_element: - _anthropic_content_element["cache_control"] = _content_element["cache_control"] - user_content.append(_anthropic_content_element) - elif m.get("type", "") == "text": - m = cast(ChatCompletionTextObject, m) - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=m["text"], - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(m), - ) - _content_element = cast(AnthropicMessagesTextParam, _content_element) - - user_content.append(_content_element) - elif m.get("type", "") == "document": - _document_content_element = cast( - AnthropicMessagesDocumentParam, - add_cache_control_to_content( - anthropic_content_element=cast(AnthropicMessagesDocumentParam, m), - original_content_element=dict(m), - ), - ) - user_content.append(_document_content_element) - elif m.get("type", "") == "file": - _file_content_element = anthropic_process_openai_file_message( - cast(ChatCompletionFileObject, m) - ) - _file_content_element = add_cache_control_to_content( - anthropic_content_element=cast( - AnthropicMessagesDocumentParam, - _file_content_element, - ), - original_content_element=dict(m), - ) - user_content.append( - cast( - AnthropicMessagesDocumentParam, - _file_content_element, - ) - ) - if ( - len(user_content) > user_content_start - and user_message_types_block.get("cache_control") is not None - and not any( - "cache_control" in content and content["cache_control"] is not None - for content in user_content[user_content_start:] - ) - ): - cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], cached_user_content - ), - original_content_element=user_message_types_block, - ) - user_content[-1] = cached_user_content - elif isinstance(user_message_types_block["content"], str): - _anthropic_content_text_element: AnthropicMessagesTextParam = { - "type": "text", - "text": user_message_types_block["content"], - } - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_content_text_element, - original_content_element=dict(user_message_types_block), - ) - - if "cache_control" in _content_element: - _anthropic_content_text_element["cache_control"] = _content_element["cache_control"] - - user_content.append(_anthropic_content_text_element) - - elif user_message_types_block["role"] == "tool" or user_message_types_block["role"] == "function": - # OpenAI's tool message content will always be a string - user_content.append( - convert_to_anthropic_tool_result(user_message_types_block, force_base64=force_base64) - ) - - msg_i += 1 - - if user_content: - new_messages.append({"role": "user", "content": user_content}) - - # Track unique tool IDs in this merge block to avoid duplication - unique_tool_ids: set[str] = set() - - assistant_content: list[AnthropicMessagesAssistantMessageValues] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_content_block: ChatCompletionAssistantMessage = messages[msg_i] - - # Extract compaction_blocks from provider_specific_fields and add them first - _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") - if isinstance(_provider_specific_fields_raw, dict): - _compaction_blocks = _provider_specific_fields_raw.get("compaction_blocks") - if _compaction_blocks and isinstance(_compaction_blocks, list): - # Add compaction blocks at the beginning of assistant content : https://platform.claude.com/docs/en/build-with-claude/compaction - assistant_content.extend(_compaction_blocks) - - _raw_thinking_blocks = assistant_content_block.get("thinking_blocks", None) - thinking_blocks = ( - _drop_unsignable_thinking_blocks(_raw_thinking_blocks) if _raw_thinking_blocks is not None else None - ) - - # Check if tool_calls contain server tool calls (web search, etc.) - # If so, we need to interleave thinking blocks with tool call groups - # to preserve the original content block ordering. - # Fixes: https://github.com/BerriAI/litellm/issues/23047 - assistant_tool_calls = assistant_content_block.get("tool_calls") - _has_server_tool_calls = False - if assistant_tool_calls is not None: - for _tc in assistant_tool_calls: - _tc_id = _tc.get("id") if isinstance(_tc, dict) else getattr(_tc, "id", None) - if _tc_id and isinstance(_tc_id, str) and _tc_id.startswith("srvtoolu_"): - _has_server_tool_calls = True - break - - _content_is_list: bool = "content" in assistant_content_block and isinstance( - assistant_content_block["content"], list - ) - _content_list = assistant_content_block.get("content") if _content_is_list else None - assistant_content_start: int = len(assistant_content) - - if ( - thinking_blocks is not None - and _has_server_tool_calls - and isinstance(assistant_content_block.get("content", None), (str, type(None))) - ): - # INTERLEAVED MODE: When we have both thinking blocks and server - # tool calls (e.g. web search), Anthropic's original response - # interleaves them: [thinking_1, server_tool_use_1, result_1, - # thinking_2, text, server_tool_use_2, result_2, ...]. - # We must preserve this interleaved order because Anthropic - # verifies thinking block signatures based on position. - - # Build the tool call groups (server_tool_use + its result) - _provider_specific_fields_raw_tc = assistant_content_block.get("provider_specific_fields") - _provider_specific_fields_tc: dict[str, Any] = {} - if isinstance(_provider_specific_fields_raw_tc, dict): - _provider_specific_fields_tc = cast(dict[str, Any], _provider_specific_fields_raw_tc) - _web_search_results_tc = _provider_specific_fields_tc.get("web_search_results") - _tool_results_tc = _provider_specific_fields_tc.get("tool_results") - tool_invoke_results = convert_to_anthropic_tool_invoke( - assistant_tool_calls, - web_search_results=_web_search_results_tc, - tool_results=_tool_results_tc, - ) - - # Group tool invoke results into (server_tool_use, result) pairs - # and separate regular tool_use blocks - server_tool_groups: list[list[Any]] = [] - regular_tool_uses: list[Any] = [] - _current_group: list[Any] = [] - for item in tool_invoke_results: - item_type = item.get("type", "") if isinstance(item, dict) else getattr(item, "type", "") - if item_type == "server_tool_use": - if _current_group: - server_tool_groups.append(_current_group) - _current_group = [item] - elif item_type.endswith("_tool_result"): - _current_group.append(item) - elif item_type == "tool_use": - regular_tool_uses.append(item) - else: - _current_group.append(item) - if _current_group: - server_tool_groups.append(_current_group) - - # Build the text block if content is a non-empty string - text_element = None - _acb_content = assistant_content_block.get("content") - if isinstance(_acb_content, str) and _acb_content: - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=_acb_content, - ) - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(assistant_content_block), - ) - if "cache_control" in _content_element: - _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] - text_element = _anthropic_text_content_element - - # Interleave: each thinking block precedes its server tool group. - # Pattern: thinking[0], group[0], thinking[1], group[1], ... - # Any remaining thinking blocks (after all groups) go before text. - # Any remaining groups (after all thinking blocks) go after. - tb_idx = 0 - grp_idx = 0 - num_tb = len(thinking_blocks) if thinking_blocks else 0 - num_grp = len(server_tool_groups) - - while tb_idx < num_tb or grp_idx < num_grp: - if tb_idx < num_tb and grp_idx < num_grp: - # Emit thinking block then its tool group - assistant_content.append(thinking_blocks[tb_idx]) - tb_idx += 1 - for block in server_tool_groups[grp_idx]: - item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) - grp_idx += 1 - elif tb_idx < num_tb: - # More thinking blocks than tool groups - emit before text - assistant_content.append(thinking_blocks[tb_idx]) - tb_idx += 1 - else: - # More tool groups than thinking blocks - emit remaining - for block in server_tool_groups[grp_idx]: - item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) - grp_idx += 1 - - # Add text block (if any) - if text_element is not None: - assistant_content.append(text_element) - - # Add regular (non-server) tool calls at the end - for item in regular_tool_uses: - item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) - if item_id and item_id in unique_tool_ids: - continue - if item_id: - unique_tool_ids.add(item_id) - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) - - # Mark tool_calls as already processed so they are not added again - assistant_tool_calls = None - - else: - # SEQUENTIAL MODE: No server tool calls, or no thinking blocks, - # or content is a list. Use the original sequential approach. - - # When content is a list, check if it already contains thinking - # blocks inline. If so, skip prepending thinking_blocks to avoid - # duplication and preserve the original interleaved order. - # Fixes the gap where list-content messages bypass INTERLEAVED - # MODE and still get thinking blocks prepended out of order. - _list_has_thinking = False - if _content_is_list and _content_list is not None: - for _item in _content_list: - if isinstance(_item, dict) and _item.get("type") in ( - "thinking", - "redacted_thinking", - ): - _list_has_thinking = True - break - - if ( - thinking_blocks is not None and not _list_has_thinking - ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR - assistant_content.extend(thinking_blocks) - if _content_is_list and _content_list is not None: - assistant_content_start = len(assistant_content) - for m in _content_list: - if not isinstance(m, dict): - continue - # handle thinking blocks - thinking_block = cast(str, m.get("thinking", "")) - text_block = cast(str, m.get("text", "")) - if ( - m.get("type", "") == "thinking" - and len(thinking_block) > 0 - and not is_unsignable_thinking_block(m) - ): # don't pass empty text blocks. anthropic api raises errors. - anthropic_message: ChatCompletionThinkingBlock | AnthropicMessagesTextParam = cast( - ChatCompletionThinkingBlock, m - ) - assistant_content.append(anthropic_message) - # handle text - elif ( - m.get("type", "") == "text" and len(text_block) > 0 - ): # don't pass empty text blocks. anthropic api raises errors. - anthropic_message = AnthropicMessagesTextParam(type="text", text=text_block) - _cached_message = add_cache_control_to_content( - anthropic_content_element=anthropic_message, - original_content_element=dict(m), - ) - - assistant_content.append(cast(AnthropicMessagesTextParam, _cached_message)) - # handle server_tool_use blocks (tool search, web search, etc.) - # Pass through as-is since these are Anthropic-native content types - elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): - assistant_content.append(m) - elif ( - "content" in assistant_content_block - and isinstance(assistant_content_block["content"], str) - and assistant_content_block["content"] # don't pass empty text blocks. anthropic api raises errors. - ): - _anthropic_text_content_element = AnthropicMessagesTextParam( - type="text", - text=assistant_content_block["content"], - ) - - _content_element = add_cache_control_to_content( - anthropic_content_element=_anthropic_text_content_element, - original_content_element=dict(assistant_content_block), - ) - - if "cache_control" in _content_element: - _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] - - assistant_content.append(_anthropic_text_content_element) - - if assistant_tool_calls is not None: # support assistant tool invoke conversion - # Get web_search_results and tool_results from provider_specific_fields - # for server_tool_use reconstruction. - # Fixes: https://github.com/BerriAI/litellm/issues/17737 - _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") - _provider_specific_fields: dict[str, Any] = {} - if isinstance(_provider_specific_fields_raw, dict): - _provider_specific_fields = cast(dict[str, Any], _provider_specific_fields_raw) - _web_search_results = _provider_specific_fields.get("web_search_results") - _tool_results = _provider_specific_fields.get("tool_results") - tool_invoke_results = convert_to_anthropic_tool_invoke( - assistant_tool_calls, - web_search_results=_web_search_results, - tool_results=_tool_results, - ) - - # Prevent "tool_use ids must be unique" errors by filtering duplicates - # This can happen when merging history that already contains the tool calls - for item in tool_invoke_results: - # tool_use items are typically dicts, but handle objects just in case - item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) - - if item_id: - if item_id in unique_tool_ids: - continue - unique_tool_ids.add(item_id) - - assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) - - assistant_function_call = assistant_content_block.get("function_call") - - if assistant_function_call is not None: - assistant_content.extend(convert_function_to_anthropic_tool_invoke(assistant_function_call)) - - if ( - _content_is_list - and len(assistant_content) > assistant_content_start - and assistant_content_block.get("cache_control") is not None - and not any( - "cache_control" in content and content["cache_control"] is not None - for content in assistant_content[assistant_content_start:] - ) - ): - cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], cached_assistant_content - ), - original_content_element=assistant_content_block, - ) - assistant_content[-1] = cached_assistant_content - - msg_i += 1 - - if assistant_content: - new_messages.append({"role": "assistant", "content": assistant_content}) - - ## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - new_messages.extend(anthropic_system_messages(messages[msg_i])) - msg_i += 1 - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - if len(new_messages) > 0 and new_messages[-1]["role"] == "assistant": - if isinstance(new_messages[-1]["content"], str): - new_messages[-1]["content"] = new_messages[-1]["content"].rstrip() - elif isinstance(new_messages[-1]["content"], list): - for content in new_messages[-1]["content"]: - if isinstance(content, dict) and content["type"] == "text": - content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message - - return new_messages - - -def extract_between_tags(tag: str, string: str, strip: bool = False) -> list[str]: - ext_list = re.findall(f"<{tag}>(.+?)", string, re.DOTALL) - if strip: - ext_list = [e.strip() for e in ext_list] - return ext_list - - -def contains_tag(tag: str, string: str) -> bool: - return bool(re.search(f"<{tag}>(.+?)", string, re.DOTALL)) - - -def parse_xml_params(xml_content, json_schema: dict | None = None): - """ - Compare the xml output to the json schema - - check if a value is a list - if so, get it's child elements - """ - root: Final = ET.fromstring(xml_content) - params: Final = {} - - if json_schema is not None: # check if we have a json schema for this function call - # iterate over all properties in the schema - for prop in json_schema["properties"]: - # If property is an array, get the nested items - _element = root.find(f"parameters/{prop}") - if json_schema["properties"][prop]["type"] == "array": - items = [] - if _element is not None: - for value in _element: - try: - if value.text is not None: - _value = json.loads(value.text) - else: - continue - except json.JSONDecodeError: - _value = value.text - items.append(_value) - params[prop] = items - # If property is not an array, append the value directly - elif _element is not None and _element.text is not None: - try: - _value = json.loads(_element.text) - except json.JSONDecodeError: - _value = _element.text - params[prop] = _value - else: - for child in root.findall(".//parameters/*"): - if child is not None and child.text is not None: - try: - # Attempt to decode the element's text as JSON - params[child.tag] = json.loads(child.text) - except json.JSONDecodeError: - # If JSON decoding fails, use the original text - params[child.tag] = child.text - - return params - - -### GEMINI HELPER FUNCTIONS ### - - -def get_system_prompt(messages): - system_prompt_indices: Final = [] - system_prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "system": - system_prompt += message["content"] - system_prompt_indices.append(idx) - if len(system_prompt_indices) > 0: - for idx in reversed(system_prompt_indices): - messages.pop(idx) - return system_prompt, messages - - -from litellm.types.llms.cohere import ( - CallObject, - ChatHistory, - ChatHistoryChatBot, - ChatHistorySystem, - ChatHistoryToolResult, - ChatHistoryUser, - ToolCallObject, - ToolResultObject, -) - - -def convert_openai_message_to_cohere_tool_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, - tool_calls: list, -) -> ToolResultObject: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "content": {"location": "San Francisco, CA", "unit": "fahrenheit", "temperature": "72"}, - }, - """ - """ - OpenAI message with a function call looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - - """ - Cohere tool_results look like: - { - "call": { - "name": "query_daily_sales_report", - "parameters": { - "day": "2023-09-29" - }, - }, - "outputs": [ - { - "date": "2023-09-29", - "summary": "Total Sales Amount: 10000, Total Units Sold: 250" - } - ] - }, - """ - - content_str: str = "" - if isinstance(message["content"], str): - content_str = message["content"] - elif isinstance(message["content"], list): - content_list: Final = message["content"] - for content in content_list: - if content["type"] == "text": - content_str += content["text"] - if len(content_str) > 0: - try: - content = json.loads(content_str) - except json.JSONDecodeError: - content = {"result": content_str} - else: - content = {} - name = "" - arguments = {} - # Recover name from last message with tool calls - if len(tool_calls) > 0: - tools: Final = tool_calls - msg_tool_call_id: Final = message.get("tool_call_id", None) - for tool in tools: - prev_tool_call_id = tool.get("id", None) - if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: - name = tool.get("function", {}).get("name", "") - arguments_str = tool.get("function", {}).get("arguments", "") - if arguments_str is not None and len(arguments_str) > 0: - arguments = json.loads(arguments_str) - - if message["role"] == "function": - function_message: Final[ChatCompletionFunctionMessage] = message - name = function_message["name"] - cohere_tool_result: ToolResultObject = { - "call": CallObject(name=name, parameters=arguments), - "outputs": [content], - } - return cohere_tool_result - else: - # We can't determine from openai message format whether it's a successful or - # error call result so default to the successful result template - - cohere_tool_result = { - "call": CallObject(name=name, parameters=arguments), - "outputs": [content], - } - return cohere_tool_result - - -def get_all_tool_calls(messages: list) -> list: - """ - Returns extracted list of `tool_calls`. - - Done to handle openai no longer returning tool call 'name' in tool results. - """ - tool_calls: Final[list] = [] - for m in messages: - if m.get("tool_calls", None) is not None: - if isinstance(m["tool_calls"], list): - tool_calls.extend(m["tool_calls"]) - - return tool_calls - - -def convert_to_cohere_tool_invoke(tool_calls: list) -> list[ToolCallObject]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - - """ - Cohere tool invokes: - { - "role": "CHATBOT", - "tool_calls": [{"name": "get_weather", "parameters": {"location": "San Francisco, CA"}}] - } - """ - - cohere_tool_invoke: Final[list[ToolCallObject]] = [ - { - "name": get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), - "parameters": json.loads(get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments")), - } - for tool in tool_calls - if get_attribute_or_key(tool, "type") == "function" - ] - - return cohere_tool_invoke - - -def cohere_messages_pt_v2( - messages: list, - model: str, - llm_provider: str, -) -> tuple[str | ToolResultObject, ChatHistory]: - """ - Returns a tuple(Union[tool_result, message], chat_history) - - - if last message is tool result -> return 'tool_result' - - if last message is text -> return message (str) - - - return preceding messages as 'chat_history' - - Note: - - cannot specify message if the last entry in chat history contains tool results - - message must be at least 1 token long or tool results must be specified. - - cannot specify tool_results if the last entry in chat history contains a user message - """ - tool_calls: Final[list] = get_all_tool_calls(messages=messages) - - ## GET MOST RECENT MESSAGE - most_recent_message: Final = messages.pop(-1) - returned_message: ToolResultObject | str = "" - if most_recent_message.get("role", "") is not None and most_recent_message["role"] == "tool": - # tool result - returned_message = convert_openai_message_to_cohere_tool_result(most_recent_message, tool_calls) - else: - content: Final[str | list] = most_recent_message.get("content") - if isinstance(content, str): - returned_message = content - else: - for chunk in content: - if chunk.get("type") == "text": - returned_message += chunk.get("text") - - ## CREATE CHAT HISTORY - user_message_types: Final = {"user"} - tool_message_types: Final = {"tool", "function"} - # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final[ChatHistory] = [] - msg_i = 0 - - while msg_i < len(messages): - user_content: str = "" - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - user_content += m["text"] - else: - user_content += messages[msg_i]["content"] - msg_i += 1 - - if len(user_content) > 0: - new_messages.append(ChatHistoryUser(role="USER", message=user_content)) - - system_content: str = "" - ## MERGE CONSECUTIVE SYSTEM CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "system": - if isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - system_content += m["text"] - else: - system_content += messages[msg_i]["content"] - msg_i += 1 - - if len(system_content) > 0: - new_messages.append(ChatHistorySystem(role="SYSTEM", message=system_content)) - - assistant_content: str = "" - assistant_tool_calls: list[ToolCallObject] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - if messages[msg_i].get("content", None) is not None and isinstance(messages[msg_i]["content"], list): - for m in messages[msg_i]["content"]: - if m.get("type", "") == "text": - assistant_content += m["text"] - elif messages[msg_i].get("content") is not None and isinstance(messages[msg_i]["content"], str): - assistant_content += messages[msg_i]["content"] - if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion - assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["tool_calls"])) - - if messages[msg_i].get("function_call"): - assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["function_call"])) - - msg_i += 1 - - if len(assistant_content) > 0: - new_messages.append( - ChatHistoryChatBot( - role="CHATBOT", - message=assistant_content, - tool_calls=assistant_tool_calls, - ) - ) - - ## MERGE CONSECUTIVE TOOL RESULTS - tool_results: list[ToolResultObject] = [] - while msg_i < len(messages) and messages[msg_i]["role"] in tool_message_types: - tool_results.append(convert_openai_message_to_cohere_tool_result(messages[msg_i], tool_calls)) - - msg_i += 1 - - if len(tool_results) > 0: - new_messages.append(ChatHistoryToolResult(role="TOOL", tool_results=tool_results)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return returned_message, new_messages - - -def cohere_message_pt(messages: list): - tool_calls: Final[list] = get_all_tool_calls(messages=messages) - prompt = "" - tool_results: Final = [] - for message in messages: - # check if this is a tool_call result - if message["role"] == "tool": - tool_result = convert_openai_message_to_cohere_tool_result(message, tool_calls=tool_calls) - tool_results.append(tool_result) - elif message.get("content"): - prompt += message["content"] + "\n\n" - prompt = prompt.rstrip() - return prompt, tool_results - - -def amazon_titan_pt( - messages: list, -): # format - https://github.com/BerriAI/litellm/issues/1896 - """ - Amazon Titan uses 'User:' and 'Bot: in it's prompt template - """ - - class AmazonTitanConstants(Enum): - HUMAN_PROMPT = "\n\nUser: " # Assuming this is similar to Anthropic prompt formatting, since amazon titan's prompt formatting is currently undocumented - AI_PROMPT = "\n\nBot: " - - prompt = "" - for idx, message in enumerate(messages): - if message["role"] == "user": - prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" - elif message["role"] == "system": - prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" - else: - prompt += f"{AmazonTitanConstants.AI_PROMPT.value}{message['content']}" - if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` - prompt = f"{AmazonTitanConstants.HUMAN_PROMPT.value}" + prompt - if messages[-1]["role"] != "assistant": - prompt += f"{AmazonTitanConstants.AI_PROMPT.value}" - return prompt - - -def _load_image_from_url(image_url): - try: - from PIL import Image - except Exception: - raise Exception("image conversion failed please run `pip install Pillow`") - from io import BytesIO - - try: - # Send a GET request to the image URL - client: Final = HTTPHandler(concurrent_limit=1) - response: Final[httpx.Response] = safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - # Check the response's content type to ensure it is an image - content_type: Final = response.headers.get("content-type") - if not content_type or "image" not in content_type: - raise ValueError(f"URL does not point to a valid image (content-type: {content_type})") - - # Load the image from the response content - return Image.open(BytesIO(response.content)) - - except Exception as e: - raise e - - -def _gemini_vision_convert_messages(messages: list): - """ - Converts given messages for GPT-4 Vision to Gemini format. - - Args: - messages (list): The messages to convert. Each message can be a dictionary with a "content" key. The content can be a string or a list of elements. If it is a string, it will be concatenated to the prompt. If it is a list, each element will be processed based on its type: - - If the element is a dictionary with a "type" key equal to "text", its "text" value will be concatenated to the prompt. - - If the element is a dictionary with a "type" key equal to "image_url", its "image_url" value will be added to the list of images. - - Returns: - tuple: A tuple containing the prompt (a string) and the processed images (a list of objects representing the images). - """ - - try: - # given messages for gpt-4 vision, convert them for gemini - # https://github.com/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_python.ipynb - prompt = "" - images: Final = [] - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - elif element["type"] == "image_url": - image_url = element["image_url"]["url"] - images.append(image_url) - # processing images passed to gemini - processed_images: Final = [] - for img in images: - if "https:/" in img: - # Case 1: Image from URL - image = _load_image_from_url(img) - processed_images.append(image) - - else: - try: - from PIL import Image - except Exception: - raise Exception("gemini image conversion failed please run `pip install Pillow`") - - if "base64" in img: - # Case 2: Base64 image data - import base64 - import io - - # Extract the base64 image data - base64_data = img.split("base64,")[1] - - # Decode the base64 image data - image_data = base64.b64decode(base64_data) - - # Load the image from the decoded data - image = Image.open(io.BytesIO(image_data)) - else: - # Case 3: Image filepath (e.g. temp.jpeg) given - image = Image.open(img) - processed_images.append(image) - content: Final = [prompt] + processed_images - return content - except Exception as e: - raise e - - -def gemini_text_image_pt(messages: list): - """ - { - "contents":[ - { - "parts":[ - {"text": "What is this picture?"}, - { - "inline_data": { - "mime_type":"image/jpeg", - "data": "'$(base64 -w0 image.jpg)'" - } - } - ] - } - ] - } - """ - try: - pass - except Exception: - raise Exception("Importing google.generativeai failed, please run 'pip install -q google-generativeai") - - prompt = "" - images: Final = [] - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - elif element["type"] == "image_url": - image_url = element["image_url"]["url"] - images.append(image_url) - - content: Final = [prompt] + images - return content - - -def azure_text_pt(messages: list): - prompt = "" - for message in messages: - if isinstance(message["content"], str): - prompt += message["content"] - elif isinstance(message["content"], list): - # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models - for element in message["content"]: - if isinstance(element, dict): - if element["type"] == "text": - prompt += element["text"] - return prompt - - -###### AZURE AI ####### -def stringify_json_tool_call_content(messages: list) -> list: - """ - - - Check 'content' in tool role -> convert to dict (if not) -> stringify - - Done for azure_ai/cohere calls to handle results of a tool call - """ - - for m in messages: - if m["role"] == "tool" and isinstance(m["content"], str): - # check if content is a valid json object - try: - json.loads(m["content"]) - except json.JSONDecodeError: - m["content"] = json.dumps({"result": m["content"]}) - - return messages - - -###### AMAZON BEDROCK ####### - -from email.message import Message - -import httpx - -from litellm.types.llms.bedrock import ( - BedrockConverseReasoningContentBlock, - BedrockConverseReasoningTextBlock, - BedrockToolSpec, - SearchResultBlock, -) -from litellm.types.llms.bedrock import ContentBlock as BedrockContentBlock -from litellm.types.llms.bedrock import DocumentBlock as BedrockDocumentBlock -from litellm.types.llms.bedrock import ImageBlock as BedrockImageBlock -from litellm.types.llms.bedrock import SourceBlock as BedrockSourceBlock -from litellm.types.llms.bedrock import ToolBlock as BedrockToolBlock -from litellm.types.llms.bedrock import ToolResultBlock as BedrockToolResultBlock -from litellm.types.llms.bedrock import ( - ToolResultContentBlock as BedrockToolResultContentBlock, -) -from litellm.types.llms.bedrock import ToolUseBlock as BedrockToolUseBlock -from litellm.types.llms.bedrock import VideoBlock as BedrockVideoBlock - - -def _parse_content_type(content_type: str) -> str: - m: Final = Message() - m["content-type"] = content_type - return m.get_content_type() - - -def _parse_mime_type(base64_data: str) -> str | None: - mime_type_match: Final = re.match(r"data:(.*?);base64", base64_data) - if mime_type_match: - return mime_type_match.group(1) - else: - return None - - -class BedrockImageProcessor: - """Handles both sync and async image processing for Bedrock conversations.""" - - @staticmethod - def _post_call_image_processing(response: httpx.Response, image_url: str = "") -> tuple[str, str]: - # Check the response's content type to ensure it is an image - content_type: str | None = response.headers.get("content-type") - - # Use helper function to infer content type with fallback logic - content_type = infer_content_type_from_url_and_content( - url=image_url, - content=response.content, - current_content_type=content_type, - ) - - content_type = _parse_content_type(content_type) - - # Convert the image content to base64 bytes - base64_bytes: Final = base64.b64encode(response.content).decode("utf-8") - - return base64_bytes, content_type - - @staticmethod - async def get_image_details_async(image_url) -> tuple[str, str]: - try: - client: Final = get_async_httpx_client( - llm_provider=httpxSpecialProvider.PromptFactory, - params={"concurrent_limit": 1}, - ) - # Send a GET request to the image URL - response: Final[httpx.Response] = await async_safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - return BedrockImageProcessor._post_call_image_processing(response, image_url) - - except Exception as e: - raise e - - @staticmethod - def get_image_details(image_url) -> tuple[str, str]: - try: - client: Final = HTTPHandler(concurrent_limit=1) - # Send a GET request to the image URL - response: Final[httpx.Response] = safe_get(client, image_url) - response.raise_for_status() # Raise an exception for HTTP errors - - return BedrockImageProcessor._post_call_image_processing(response, image_url) - - except Exception as e: - raise e - - @staticmethod - def _parse_base64_image(image_url: str) -> tuple[str, str, str]: - """Parse base64 encoded image data.""" - image_metadata, img_without_base_64 = image_url.split(",") - - # Extract MIME type using regular expression - mime_type_match: Final = re.match(r"data:(.*?);base64", image_metadata) - - if mime_type_match: - mime_type = mime_type_match.group(1) - mime_type = mime_type.split(";")[0] - image_format = mime_type.split("/")[1] - else: - mime_type = "image/jpeg" - image_format = "jpeg" - - return img_without_base_64, mime_type, image_format - - @staticmethod - def _validate_format(mime_type: str, image_format: str) -> str: - """Validate image format and mime type for both images and documents.""" - - supported_image_formats: Final = litellm.AmazonConverseConfig().get_supported_image_types() - supported_doc_formats: Final = litellm.AmazonConverseConfig().get_supported_document_types() - supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() - - document_types: Final = ["application", "text"] - is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) - - supported_image_and_video_formats: Final[list[str]] = supported_video_formats + supported_image_formats - - if is_document: - return BedrockImageProcessor._get_document_format( - mime_type=mime_type, supported_doc_formats=supported_doc_formats - ) - - else: - ######################################################### - # Check if image_format is an image or video - ######################################################### - if image_format not in supported_image_and_video_formats: - raise ValueError( - f"Unsupported image format: {image_format}. Supported formats: {supported_image_and_video_formats}" - ) - return image_format - - @staticmethod - def _get_document_format(mime_type: str, supported_doc_formats: list[str]) -> str: - """ - Get the document format from the mime type - - - Primary method - uses `mimetypes.guess_all_extensions` - - Fallback method - uses `get_file_extension_from_mime_type` - - Relevant Issue: https://github.com/BerriAI/litellm/issues/12260 - - `mimetypes` is not available in docker containers, so we fallback to `get_file_extension_from_mime_type` - - Args: - mime_type: The mime type of the document - supported_doc_formats: The supported document formats for the current model - - Returns: - The document format - """ - valid_extensions: list[str] | None = None - potential_extensions: Final = mimetypes.guess_all_extensions(mime_type, strict=False) - valid_extensions = [ext[1:] for ext in potential_extensions if ext[1:] in supported_doc_formats] - - # Fallback to types/files.py if mimetypes doesn't return valid extensions - ################# - # litellm runs on docker containers and `mimetypes` depends on the installed mimetypes of the OS - # we fallback to well known mime types in types/files.py if mimetypes doesn't return valid extensions - if not valid_extensions: - try: - fallback_extension: Final = get_file_extension_from_mime_type(mime_type) - if fallback_extension in supported_doc_formats: - valid_extensions = [fallback_extension] - except ValueError: - # Neither mimetypes nor files.py could handle this MIME type - # get_file_extension_from_mime_type raises ValueError if the mime type is not supported - pass - - if not valid_extensions: - raise ValueError( - f"No supported extensions for MIME type: {mime_type}. Supported formats: {supported_doc_formats}" - ) - - # Use first valid extension instead of provided image_format - return valid_extensions[0] - - @staticmethod - def _create_bedrock_block(image_bytes: str, mime_type: str, image_format: str) -> BedrockContentBlock: - """Create appropriate Bedrock content block based on mime type.""" - _blob: Final = BedrockSourceBlock(bytes=image_bytes) - - document_types: Final = ["application", "text"] - is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) - - supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() - is_video: Final = any(image_format.startswith(video_type) for video_type in supported_video_formats) - - HASH_SAMPLE_BYTES: Final = 64 * 1024 # hash up to 64 KB of data - - if is_document: - # --- Prepare normalized bytes for hashing (without modifying original) --- - if isinstance(image_bytes, str): - # Remove whitespace/newlines so base64 variations hash identically - normalized = "".join(image_bytes.split()).encode("utf-8") - else: - normalized = image_bytes - - # --- Use only the first 64 KB for speed --- - if len(normalized) <= HASH_SAMPLE_BYTES: - sample = normalized - else: - sample = normalized[:HASH_SAMPLE_BYTES] - - # --- Compute deterministic hash (sample + total length) --- - hasher: Final = hashlib.sha256() - hasher.update(sample) - hasher.update(str(len(normalized)).encode("utf-8")) # include full length for uniqueness - full_hash: Final = hasher.hexdigest() - content_hash: Final = full_hash[:16] # short deterministic ID - - document_name: Final = f"DocumentPDFmessages_{content_hash}_{image_format}" - - # --- Return content block --- - return BedrockContentBlock( - document=BedrockDocumentBlock( - source=_blob, - format=image_format, - name=document_name, - ) - ) - elif is_video: - return BedrockContentBlock(video=BedrockVideoBlock(source=_blob, format=image_format)) - else: - return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) - - @classmethod - def process_image_sync(cls, image_url: str, format: str | None = None) -> BedrockContentBlock: - """Synchronous image processing.""" - - if "base64" in image_url: - img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) - elif "http://" in image_url or "https://" in image_url: - img_bytes, mime_type = BedrockImageProcessor.get_image_details(image_url) - image_format = mime_type.split("/")[1] - else: - raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") - - if format: - mime_type = format - image_format = mime_type.split("/")[1] - - image_format = cls._validate_format(mime_type, image_format) - return cls._create_bedrock_block(img_bytes, mime_type, image_format) - - @classmethod - async def process_image_async(cls, image_url: str, format: str | None) -> BedrockContentBlock: - """Asynchronous image processing.""" - - if "base64" in image_url: - img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) - elif "http://" in image_url or "https://" in image_url: - img_bytes, mime_type = await BedrockImageProcessor.get_image_details_async(image_url) - image_format = mime_type.split("/")[1] - else: - raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") - - if format: # override with user-defined params - mime_type = format - image_format = mime_type.split("/")[1] - - image_format = cls._validate_format(mime_type, image_format) - return cls._create_bedrock_block(img_bytes, mime_type, image_format) - - -def _convert_to_bedrock_tool_call_invoke( - tool_calls: list, - model: str | None = None, -) -> list[BedrockContentBlock]: - """ - OpenAI tool invokes: - { - "role": "assistant", - "content": null, - "tool_calls": [ - { - "id": "call_abc123", - "type": "function", - "function": { - "name": "get_current_weather", - "arguments": "{\n\"location\": \"Boston, MA\"\n}" - } - } - ] - }, - """ - """ - Bedrock tool invokes: - [ - { - "role": "assistant", - "toolUse": { - "input": {"location": "Boston, MA", ..}, - "name": "get_current_weather", - "toolUseId": "call_abc123" - } - } - ] - """ - """ - - json.loads argument - - extract name - - extract id - """ - from litellm.litellm_core_utils.prompt_templates.common_utils import ( - split_concatenated_json_objects, - ) - - try: - _parts_list: Final[list[BedrockContentBlock]] = [] - for tool in tool_calls: - if "function" in tool: - tool_id = tool["id"] - name = make_valid_bedrock_tool_name(tool["function"].get("name", "")) - arguments = tool["function"].get("arguments", "") - - if not arguments or not arguments.strip(): - arguments_dict = {} - else: - try: - arguments_dict = json.loads(arguments) - # Ensure arguments_dict is always a dict - # (Bedrock requires toolUse.input to be an object). - # Some providers return arguments: '""' which - # json.loads decodes to a bare string. - if not isinstance(arguments_dict, dict): - arguments_dict = {} - except json.JSONDecodeError: - # The model may return multiple JSON objects - # concatenated in a single arguments string, e.g. - # '{"cmd":"a"}{"cmd":"b"}{"cmd":"c"}' - # Split them and emit one toolUse block per object. - # Fixes: https://github.com/BerriAI/litellm/issues/20543 - parsed_objects = split_concatenated_json_objects(arguments) - if parsed_objects: - # First object keeps the original tool id. - for obj_idx, obj in enumerate(parsed_objects): - block_id = _sanitize_bedrock_tool_use_id( - tool_id if obj_idx == 0 else f"{tool_id}_{obj_idx}" - ) - bedrock_tool = BedrockToolUseBlock(input=obj, name=name, toolUseId=block_id) - _parts_list.append(BedrockContentBlock(toolUse=bedrock_tool)) - # cache_control applies to the whole original - # tool call; attach after the last split block. - if tool.get("cache_control", None) is not None: - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool["cache_control"]}, - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts_list.append(_cache_point_block) - continue - # Fallback: no objects extracted — use empty dict. - arguments_dict = {} - - bedrock_tool = BedrockToolUseBlock( - input=arguments_dict, name=name, toolUseId=_sanitize_bedrock_tool_use_id(tool_id) - ) - bedrock_content_block = BedrockContentBlock(toolUse=bedrock_tool) - _parts_list.append(bedrock_content_block) - - # Check for cache_control and add a separate cachePoint block - if tool.get("cache_control", None) is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool["cache_control"]}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - _parts_list.append(cache_point_block) - return _parts_list - except Exception as e: - tool_call_ids: Final = tuple(tool.get("id") for tool in tool_calls if isinstance(tool, dict)) - raise litellm.BadRequestError( - message=f"Unable to convert openai tool calls with ids={tool_call_ids} to bedrock tool calls. " - f"Received error={e}", - model=model or "", - llm_provider="bedrock", - ) from e - - -def _append_bedrock_tool_result_media_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - processed_block: BedrockContentBlock, - content: dict, - content_type: str, -) -> None: - if "image" in processed_block: - tool_result_content_blocks.append(BedrockToolResultContentBlock(image=processed_block["image"])) - elif "document" in processed_block: - tool_result_content_blocks.append(BedrockToolResultContentBlock(document=processed_block["document"])) - else: - verbose_logger.warning( - "Bedrock Converse: unrecognized BedrockContentBlock keys %s for %s tool-result block %s; dropping.", - list(processed_block.keys()), - content_type, - content, - ) - - -def _append_bedrock_tool_result_image_url_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - content: dict, -) -> None: - format: str | None = None - if isinstance(content["image_url"], dict): - image_url = content["image_url"]["url"] - format = content["image_url"].get("format") - else: - image_url = content["image_url"] - processed_block: Final = BedrockImageProcessor.process_image_sync( - image_url=image_url, - format=format, - ) - _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "image_url") - - -def _append_bedrock_tool_result_file_block( - tool_result_content_blocks: list[BedrockToolResultContentBlock], - content: dict, -) -> None: - # Match the user-message path (_process_file_message): accept either - # file_data (base64 data URI) or file_id (server-side reference / URL). - file_obj: Final = content.get("file") or {} - file_data: Final = file_obj.get("file_data") - file_id: Final = file_obj.get("file_id") - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={content}", - model="", - llm_provider="bedrock", - ) - processed_block: Final = BedrockImageProcessor.process_image_sync( - image_url=cast(str, file_id or file_data), - format=file_obj.get("format"), - ) - _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "file") - - -def _parse_bedrock_tool_result_content_list( - content_list: list, -) -> list[BedrockToolResultContentBlock]: - tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] - for content in content_list: - if content["type"] == "text": - tool_result_content_blocks.append(BedrockToolResultContentBlock(text=content["text"])) - elif content["type"] == "image_url": - _append_bedrock_tool_result_image_url_block(tool_result_content_blocks, content) - elif content["type"] == "file": - _append_bedrock_tool_result_file_block(tool_result_content_blocks, content) - return tool_result_content_blocks - - -def _build_bedrock_tool_result_content_blocks( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, -) -> tuple[list[BedrockToolResultContentBlock], bool]: - # Optional OpenAI tool-message extension: - # allow structured Bedrock search results on tool messages and map them - # directly to toolResult.content[].searchResult for Converse API. - # - # If `search_results` is present, we intentionally prefer it over `content` - # to avoid generating mixed text + searchResult blocks. - search_results: Final = message.get("search_results") - if isinstance(search_results, list): - tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] - for result in search_results: - if not isinstance(result, dict): - continue - tool_result_content_blocks.append( - BedrockToolResultContentBlock(searchResult=cast(SearchResultBlock, result)) - ) - if tool_result_content_blocks: - return tool_result_content_blocks, True - - message_content: Final = message["content"] - if isinstance(message_content, str): - return [BedrockToolResultContentBlock(text=message_content)], False - if isinstance(message_content, list): - return _parse_bedrock_tool_result_content_list(message_content), False - return [], False - - -def _convert_to_bedrock_tool_call_result( - message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, -) -> BedrockContentBlock: - """ - OpenAI message with a tool result looks like: - { - "tool_call_id": "tool_1", - "role": "tool", - "name": "get_current_weather", - "content": "function result goes here", - }, - - OpenAI message with a function call result looks like: - { - "role": "function", - "name": "get_current_weather", - "content": "function result goes here", - } - """ - """ - Bedrock result looks like this: - { - "role": "user", - "content": [ - { - "toolResult": { - "toolUseId": "tooluse_kZJMlvQmRJ6eAyJE5GIl7Q", - "content": [ - { - "json": { - "song": "Elemental Hotel", - "artist": "8 Storey Hike" - } - } - ] - } - } - ] - } - """ - """ - - - """ - tool_result_content_blocks, used_search_results = _build_bedrock_tool_result_content_blocks(message) - - message.get("name", "") - id: Final = _sanitize_bedrock_tool_use_id(str(message.get("tool_call_id", str(uuid.uuid4())))) - - tool_result: Final = BedrockToolResultBlock(content=tool_result_content_blocks, toolUseId=id) - if used_search_results: - tool_result["status"] = cast(Literal["success"], "success") - - content_block: Final = BedrockContentBlock(toolResult=tool_result) - - return content_block - - -def _deduplicate_bedrock_content_blocks( - blocks: list[BedrockContentBlock], - block_key: str, - id_key: str = "toolUseId", -) -> list[BedrockContentBlock]: - """ - Remove duplicate content blocks that share the same ID under ``block_key``. - - Bedrock requires all toolResult and toolUse IDs within a single message to - be unique. When merging consecutive messages, duplicates can occur if the - same tool_call_id appears multiple times in conversation history. - - When duplicates exist, the first occurrence is retained and subsequent ones - are discarded. A warning is logged for every dropped block so that - upstream duplication bugs remain visible. - - Blocks that do not contain ``block_key`` (e.g., cachePoint, text) are - always preserved. - - Args: - blocks: The list of Bedrock content blocks to deduplicate. - block_key: The dict key to inspect (e.g. ``"toolResult"`` or ``"toolUse"``). - id_key: The nested key that holds the unique ID (default ``"toolUseId"``). - """ - seen_ids: Final[set[str]] = set() - deduplicated: Final[list[BedrockContentBlock]] = [] - for block in blocks: - keyed = block.get(block_key) - if keyed is not None and isinstance(keyed, dict): - block_id = keyed.get(id_key) - if block_id: - if block_id in seen_ids: - verbose_logger.warning( - "Bedrock Converse: dropping duplicate %s block with " - "%s=%s. This may indicate duplicate tool messages in " - "conversation history.", - block_key, - id_key, - block_id, - ) - continue - seen_ids.add(block_id) - deduplicated.append(block) - return deduplicated - - -def _deduplicate_bedrock_tool_content( - tool_content: list[BedrockContentBlock], -) -> list[BedrockContentBlock]: - """Convenience wrapper: deduplicate ``toolResult`` blocks by ``toolUseId``.""" - return _deduplicate_bedrock_content_blocks(tool_content, "toolResult") - - -def _rename_duplicate_bedrock_document_names( - contents: list[BedrockMessageBlock], -) -> list[BedrockMessageBlock]: - """ - Rename duplicate document names across all messages in a Bedrock request. - - Document names are derived from a content hash, so the same file appearing - in multiple conversation turns produces identical names and Bedrock rejects - the request with "Messages can not contain duplicate document names". The - first occurrence keeps its original name so prompt-cache prefixes stay - stable; later occurrences get a deterministic positional suffix - (``_2``, ``_3``, ...), bumped further if the suffixed name already - belongs to another document (e.g. an organic name ending in ``_2``). - """ - used_names: Final[set[str]] = set() - for message in contents: - for block in message.get("content") or []: - document = block.get("document") - if isinstance(document, dict) and document.get("name"): - used_names.add(document["name"]) - - name_counts: Final[dict[str, int]] = {} - for message in contents: - for block in message.get("content") or []: - document = block.get("document") - if not isinstance(document, dict): - continue - name = document.get("name") - if not name: - continue - count = name_counts.get(name, 0) + 1 - name_counts[name] = count - if count > 1: - suffix = count - new_name = f"{name}_{suffix}" - while new_name in used_names: - suffix += 1 - new_name = f"{name}_{suffix}" - used_names.add(new_name) - document["name"] = new_name - return contents - - -BEDROCK_DOCUMENT_PLACEHOLDER_TEXT: Final = "." - - -def _with_text_when_document_only(message: BedrockMessageBlock) -> BedrockMessageBlock: - blocks: Final = message["content"] - needs_text: Final = ( - message["role"] == "user" - and any("document" in block for block in blocks) - and all("text" not in block for block in blocks) - ) - if not needs_text: - return message - placeholder: Final = BedrockContentBlock(text=BEDROCK_DOCUMENT_PLACEHOLDER_TEXT) - cut: Final = len(blocks) - 1 if "cachePoint" in blocks[-1] else len(blocks) - return BedrockMessageBlock(role="user", content=[*blocks[:cut], placeholder, *blocks[cut:]]) - - -def _ensure_document_messages_have_text( - contents: list[BedrockMessageBlock], -) -> list[BedrockMessageBlock]: - """ - Bedrock Converse rejects any user message that carries a document block - without a sibling text block ("A text block must be included when using - documents"), e.g. Claude Code sends the PDF as a document-only user turn. - Inject a placeholder text block, kept ahead of a trailing cachePoint so - the caller's cache boundary stays the final block. - """ - return [_with_text_when_document_only(message) for message in contents] - - -def _sort_bedrock_assistant_content_blocks( - blocks: list[BedrockContentBlock], -) -> list[BedrockContentBlock]: - """ - Sort assistant content blocks so that ``text`` blocks appear before - ``toolUse`` blocks. - - Bedrock requires all ``text`` blocks to precede any ``toolUse`` blocks - within an assistant message. When the Responses API converts - function_call items before message items, the resulting ``toolUse`` - blocks can end up before ``text`` blocks, causing Bedrock to reject - the request with a 400 error because the ``toolUse`` → ``toolResult`` - pairing is broken by the intervening ``text`` block. - - Sort order (stable): - 0 - reasoningContent - 1 - text / image / document / video / other non-tool blocks - 2 - toolUse - """ - - def _sort_key(block: BedrockContentBlock) -> int: - if "reasoningContent" in block: - return 0 - if "toolUse" in block: - return 2 - if "cachePoint" in block: - # cachePoint blocks are paired with their preceding toolUse block. - # Same key as toolUse so Python's stable sort keeps them together. - return 2 - return 1 - - return sorted(blocks, key=_sort_key) - - -def _insert_assistant_continue_message( - messages: list[BedrockMessageBlock], - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> list[BedrockMessageBlock]: - """ - Add dummy message between user/tool result blocks. - - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 - """ - if assistant_continue_message is not None: - if isinstance(assistant_continue_message, str): - messages.append( - BedrockMessageBlock( - role="assistant", - content=[BedrockContentBlock(text=assistant_continue_message)], - ) - ) - elif isinstance(assistant_continue_message, dict): - text = convert_content_list_to_str(assistant_continue_message) - messages.append( - BedrockMessageBlock( - role="assistant", - content=[BedrockContentBlock(text=text)], - ) - ) - elif litellm.modify_params: - text = convert_content_list_to_str(cast(ChatCompletionAssistantMessage, DEFAULT_ASSISTANT_CONTINUE_MESSAGE)) - messages.append( - BedrockMessageBlock( - role="assistant", - content=[ - BedrockContentBlock(text=text), - ], - ) - ) - return messages - - -def get_user_message_block_or_continue_message( - message: ChatCompletionUserMessage, - user_continue_message: ChatCompletionUserMessage | None = None, -) -> ChatCompletionUserMessage: - """ - Returns the user content block - if content block is an empty string, then return the default continue message - - Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 - """ - content_block: Final = message.get("content", None) - - # Handle None case - if content_block is None or (user_continue_message is None and litellm.modify_params is False): - return skip_empty_text_blocks(message=message) - - # Handle string case - if isinstance(content_block, str): - # check if content is empty - if content_block.strip(): - return message - else: - return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) - - # Handle list case - if isinstance(content_block, list): - """ - CHECK FOR - "content": [ - { - "type": "text", - "text": "" - } - ], - """ - if not content_block: - return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) - # Create a copy of the message to avoid modifying the original - modified_content_block: Final = content_block.copy() - - for item in modified_content_block: - # Check if the list is empty - if item["type"] == "text": - if not item["text"].strip(): - # Replace empty text with continue message - _user_continue_message = ChatCompletionUserMessage( - **(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE) - ) - text = convert_content_list_to_str(_user_continue_message) - item["text"] = text - break - modified_message: Final = message.copy() - modified_message["content"] = modified_content_block - return modified_message - - # Handle unsupported type - raise ValueError(f"Unsupported content type: {type(content_block)}") - - -def return_assistant_continue_message( - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - if assistant_continue_message and isinstance(assistant_continue_message, str): - return ChatCompletionAssistantMessage( - role="assistant", - content=assistant_continue_message, - ) - elif assistant_continue_message and isinstance(assistant_continue_message, dict): - return ChatCompletionAssistantMessage(**assistant_continue_message) - else: - return DEFAULT_ASSISTANT_CONTINUE_MESSAGE - - -def _skip_empty_dict_blocks(blocks: list[dict]) -> list[dict]: - """ - Filter out empty text blocks from a list of dictionaries. - - Args: - blocks: List of dictionaries representing message content blocks - - Returns: - Filtered list of non-empty text blocks - """ - return [item for item in blocks if not (item.get("type") == "text" and not item.get("text", "").strip())] - - -@overload -def skip_empty_text_blocks( - message: ChatCompletionAssistantMessage, -) -> ChatCompletionAssistantMessage: - pass - - -@overload -def skip_empty_text_blocks( - message: ChatCompletionUserMessage, -) -> ChatCompletionUserMessage: - pass - - -def skip_empty_text_blocks( - message: ChatCompletionAssistantMessage | ChatCompletionUserMessage, -) -> ChatCompletionAssistantMessage | ChatCompletionUserMessage: - """ - Skips empty text blocks in message content text blocks. - - Do not insert content here. This is a helper function, which can also be used in base case. - """ - content_block: Final = message.get("content", None) - if content_block is None: - return message - if ( - isinstance(content_block, str) - and not content_block.strip() - and is_non_content_values_set(message) - and message["role"] == "assistant" - ): - modified_message = message.copy() - modified_message["content"] = None # user message content cannot be None - return modified_message - elif isinstance(content_block, list): - modified_content_block: Final = _skip_empty_dict_blocks(cast(list[dict], content_block)) - - # If no content remains and it's an assistant message, set content to None - if not modified_content_block and message["role"] == "assistant": - modified_message = message.copy() - modified_message["content"] = None - return modified_message - - modified_message_alt: Final = message.copy() - - # Type-specific casting based on message role - if message["role"] == "assistant": - modified_message_alt["content"] = cast( - list[OpenAIMessageContentListBlock] | None, - modified_content_block or None, - ) - elif message["role"] == "user" and modified_content_block is not None: - modified_message_alt["content"] = cast(list[ChatCompletionTextObject] | None, modified_content_block) - - return modified_message_alt - - return message - - -def process_empty_text_blocks( - message: ChatCompletionAssistantMessage, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - modified_content_block = message.get("content", None) - ## BASE CASE ## - if modified_content_block is None or not isinstance(modified_content_block, list): - return message - - # Check if all items are empty text blocks - if all(item["type"] == "text" and not item["text"].strip() for item in modified_content_block): - # Replace with a single continue message - _assistant_continue_message: Final = return_assistant_continue_message(assistant_continue_message) - modified_content_block = [ - { - "type": "text", - "text": convert_content_list_to_str(_assistant_continue_message), - } - ] - else: - # Filter out only empty text blocks, keeping non-empty text and other block types - modified_content_block = [ - item for item in modified_content_block if not (item["type"] == "text" and not item["text"].strip()) - ] - - modified_message: Final = message.copy() - modified_message["content"] = cast( - list[ChatCompletionTextObject] | list[ChatCompletionThinkingBlock], - modified_content_block, - ) - return modified_message - - -def get_assistant_message_block_or_continue_message( - message: ChatCompletionAssistantMessage, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> ChatCompletionAssistantMessage: - """ - Returns the user content block - if content block is an empty string, then return the default continue message - - Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 - """ - content_block: Final = message.get("content", None) - - # Handle Base case - if content_block is None or (assistant_continue_message is None and litellm.modify_params is False): - return skip_empty_text_blocks(message=message) - - # Handle string case - if isinstance(content_block, str): - # check if content is empty - if content_block.strip(): - return message - else: - if is_non_content_values_set(message): - modified_message: Final = message.copy() - modified_message["content"] = None - return modified_message - return return_assistant_continue_message(assistant_continue_message) - - # Handle list case - if isinstance(content_block, list): - """ - CHECK FOR - "content": [ - { - "type": "text", - "text": "" - } - ], - """ - return process_empty_text_blocks(message=message, assistant_continue_message=assistant_continue_message) - - # Handle unsupported type - raise ValueError(f"Unsupported content type: {type(content_block)}") - - -class BedrockConverseMessagesProcessor: - @staticmethod - def _initial_message_setup( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - ) -> list: - # gracefully handle base case of no messages at all - if len(messages) == 0: - if user_continue_message is not None: - messages.append(user_continue_message) - elif litellm.modify_params: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE) - else: - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + "bedrock requires at least one non-system message", - model=model, - llm_provider=llm_provider, - ) - - # if initial message is assistant message - if messages[0].get("role") is not None and messages[0]["role"] == "assistant": - if not messages[0].get("prefix"): - if user_continue_message is not None: - messages.insert(0, user_continue_message) - elif litellm.modify_params: - messages.insert(0, DEFAULT_USER_CONTINUE_MESSAGE) - - # if final message is assistant message - if messages[-1].get("role") is not None and messages[-1]["role"] == "assistant": - if not messages[-1].get("prefix"): - if user_continue_message is not None: - messages.append(user_continue_message) - elif litellm.modify_params: - messages.append(DEFAULT_USER_CONTINUE_MESSAGE) - return messages - - @staticmethod - async def _bedrock_converse_messages_pt_async( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, - ) -> list[BedrockMessageBlock]: - contents: list[BedrockMessageBlock] = [] - msg_i = 0 - - messages = BedrockConverseMessagesProcessor._initial_message_setup( - messages, model, llm_provider, user_continue_message - ) - - while msg_i < len(messages): - user_content: list[BedrockContentBlock] = [] - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "user": - message_block = get_user_message_block_or_continue_message( - message=messages[msg_i], - user_continue_message=user_continue_message, - ) - if isinstance(message_block["content"], list): - _parts: list[BedrockContentBlock] = [] - for element in message_block["content"]: - if isinstance(element, dict): - if element["type"] == "text": - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "guarded_text": - # Wrap guarded_text in guardContent block - _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) - _parts.append(_part) - elif element["type"] in ("grounding_source", "query"): - # Contextual grounding tags are guardrail metadata; the - # model only needs the underlying text, so render them - # as plain text on the generate path. - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "image_url": - format: str | None = None - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - format = element["image_url"].get("format") - else: - image_url = element["image_url"] - _part = await BedrockImageProcessor.process_image_async( - image_url=image_url, format=format - ) - _parts.append(_part) - elif element["type"] == "file": - _part = await BedrockConverseMessagesProcessor._async_process_file_message( - message=cast(ChatCompletionFileObject, element) - ) - _parts.append(_part) - elif element["type"] == "document": - _part = BedrockConverseMessagesProcessor._process_document_message(element) - _parts.append(_part) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - if ( - _parts - and message_block.get("cache_control") is not None - and not any("cachePoint" in part for part in _parts) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - user_content.extend(_parts) - elif message_block["content"] and isinstance(message_block["content"], str): - _part = BedrockContentBlock(text=messages[msg_i]["content"]) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - user_content.append(_part) - if _cache_point_block is not None: - user_content.append(_cache_point_block) - - msg_i += 1 - if user_content: - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=user_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(user_content) - else: - contents.append(BedrockMessageBlock(role="user", content=user_content)) - - ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## - tool_content: list[BedrockContentBlock] = [] - while msg_i < len(messages) and messages[msg_i]["role"] == "tool": - current_message = messages[msg_i] - tool_call_result = _convert_to_bedrock_tool_call_result(current_message) - tool_content.append(tool_call_result) - - # Check if we need to add a separate cachePoint block - tool_msg_cache_control = None - - # Check for message-level cache_control - if current_message.get("cache_control", None) is not None: - tool_msg_cache_control = current_message["cache_control"] - # Check for content-level cache_control in list content - elif isinstance(current_message.get("content"), list): - for content_element in current_message["content"]: - if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: - tool_msg_cache_control = content_element["cache_control"] - break - - # Add a separate cachePoint block if cache_control is present - if tool_msg_cache_control is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool_msg_cache_control}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - tool_content.append(cache_point_block) - - msg_i += 1 - # Deduplicate toolResult blocks with the same toolUseId - tool_content = _deduplicate_bedrock_tool_content(tool_content) - if tool_content: - # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(tool_content) - else: - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - assistant_content: list[BedrockContentBlock] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_message_block = get_assistant_message_block_or_continue_message( - message=messages[msg_i], - assistant_continue_message=assistant_continue_message, - ) - assistant_message_content_start: int = len(assistant_content) - _assistant_content = assistant_message_block.get("content", None) - thinking_blocks = cast( - list[ChatCompletionThinkingBlock] | None, - assistant_message_block.get("thinking_blocks"), - ) - - if thinking_blocks is not None: - converted_thinking_blocks = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks - ) - ) - assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=converted_thinking_blocks, - assistant_parts=assistant_content, - ) - - if _assistant_content is not None and isinstance(_assistant_content, list): - assistants_parts: list[BedrockContentBlock] = [] - for element in _assistant_content: - if isinstance(element, dict): - if element["type"] == "thinking": - thinking_block = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] - ) - assistants_parts = ( - BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=thinking_block, - assistant_parts=assistants_parts, - ) - ) - elif element["type"] == "text": - # Skip completely empty strings to avoid blank content blocks - if element.get("text", "").strip(): - assistants_part = BedrockContentBlock(text=element["text"]) - assistants_parts.append(assistants_part) - elif element["type"] == "image_url": - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - else: - image_url = element["image_url"] - assistants_part = await BedrockImageProcessor.process_image_async(image_url=image_url) - assistants_parts.append(assistants_part) - # Add cache point block for assistant content elements - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - assistants_parts.append(_cache_point_block) - assistant_content.extend(assistants_parts) - elif _assistant_content is not None and isinstance(_assistant_content, str): - # Skip completely empty strings to avoid blank content blocks - if _assistant_content.strip(): - assistant_content.append(BedrockContentBlock(text=_assistant_content)) - # If content is empty/whitespace, skip it (don't add a placeholder) - # Add cache point block for assistant string content - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - _tool_calls = assistant_message_block.get("tool_calls", []) - if _tool_calls: - assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) - - if ( - isinstance(_assistant_content, list) - and len(assistant_content) > assistant_message_content_start - and assistant_message_block.get("cache_control") is not None - and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - msg_i += 1 - - assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") - assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) - - if assistant_content: - contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) - - @staticmethod - def translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks: list[ChatCompletionThinkingBlock], - ) -> list[BedrockContentBlock]: - reasoning_content_blocks: Final[list[BedrockContentBlock]] = [] - for thinking_block in thinking_blocks: - reasoning_text = thinking_block.get("thinking") - reasoning_signature = thinking_block.get("signature") - text_block = BedrockConverseReasoningTextBlock( - text=reasoning_text or "", - ) - if reasoning_signature is not None: - text_block["signature"] = reasoning_signature - reasoning_content_block = BedrockConverseReasoningContentBlock( - reasoningText=text_block, - ) - bedrock_content_block = BedrockContentBlock(reasoningContent=reasoning_content_block) - reasoning_content_blocks.append(bedrock_content_block) - return reasoning_content_blocks - - @staticmethod - def _process_file_message(message: ChatCompletionFileObject) -> BedrockContentBlock: - file_message: Final = message.get("file") - if file_message is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="bedrock", - ) - file_data: Final = file_message.get("file_data") - file_id: Final = file_message.get("file_id") - - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={message}", - model="", - llm_provider="bedrock", - ) - format: Final = file_message.get("format") - return BedrockImageProcessor.process_image_sync(image_url=cast(str, file_id or file_data), format=format) - - @staticmethod - async def _async_process_file_message( - message: ChatCompletionFileObject, - ) -> BedrockContentBlock: - file_message: Final = message.get("file") - if file_message is None: - raise litellm.BadRequestError( - message="Content block has type='file' but is missing the required 'file' field", - model=None, - llm_provider="bedrock", - ) - file_data: Final = file_message.get("file_data") - file_id: Final = file_message.get("file_id") - format: Final = file_message.get("format") - if file_data is None and file_id is None: - raise litellm.BadRequestError( - message=f"file_data and file_id cannot both be None. Got={message}", - model="", - llm_provider="bedrock", - ) - return await BedrockImageProcessor.process_image_async(image_url=cast(str, file_id or file_data), format=format) - - @staticmethod - def _process_document_message(element: dict) -> BedrockContentBlock: - """Convert a document content block to a Bedrock DocumentBlock. - - Handles the Anthropic-style document format: - {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "..."}} - """ - source: Final = element["source"] - source_type: Final = source.get("type") - if source_type != "base64": - raise ValueError( - f"Bedrock Converse only supports base64-encoded document sources, got '{source_type}'. " - "Please convert the document to base64 before sending to Bedrock." - ) - media_type: Final[str] = source["media_type"] - data: Final[str] = source["data"] - doc_format = BedrockImageProcessor._validate_format(mime_type=media_type, image_format=media_type.split("/")[1]) - - # Deterministic name using the same hashing pattern as _create_bedrock_block - HASH_SAMPLE_BYTES: Final = 64 * 1024 - normalized: Final = "".join(data.split()).encode("utf-8") - sample: Final = normalized[:HASH_SAMPLE_BYTES] - hasher: Final = hashlib.sha256() - hasher.update(sample) - hasher.update(str(len(normalized)).encode("utf-8")) - content_hash: Final = hasher.hexdigest()[:16] - document_name: Final = f"Document_{content_hash}_{doc_format}" - - return BedrockContentBlock( - document=BedrockDocumentBlock( - source=BedrockSourceBlock(bytes=data), - format=doc_format, - name=document_name, - ) - ) - - @staticmethod - def add_thinking_blocks_to_assistant_content( - thinking_blocks: list[BedrockContentBlock], - assistant_parts: list[BedrockContentBlock], - ) -> list[BedrockContentBlock]: - """ - If contains 'signature', it is a thinking block. - If missing 'signature', it is a text block - e.g. when using a non-anthropic model. - - Handle error raised by bedrock if thinking blocks are provided for a non-thinking model (e.g. nova with tool use) - - Relevant Issue: https://github.com/BerriAI/litellm/issues/9063 - """ - filtered_thinking_blocks: Final = [] - for block in thinking_blocks: - reasoning_content = block.get("reasoningContent", None) - reasoning_text = reasoning_content.get("reasoningText", None) if reasoning_content is not None else None - if reasoning_text and not reasoning_text.get("signature"): - reasoning_text_text = reasoning_text["text"] - if reasoning_text_text.strip(): - assistants_part = BedrockContentBlock(text=reasoning_text_text) - assistant_parts.append(assistants_part) - else: - filtered_thinking_blocks.append(block) - if len(filtered_thinking_blocks) > 0: - assistant_parts.extend(filtered_thinking_blocks) - return assistant_parts - - -def _bedrock_converse_messages_pt( - messages: list, - model: str, - llm_provider: str, - user_continue_message: ChatCompletionUserMessage | None = None, - assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, -) -> list[BedrockMessageBlock]: - """ - Converts given messages from OpenAI format to Bedrock format - - - Roles must alternate b/w 'user' and 'model' (same as anthropic -> merge consecutive roles) - - Please ensure that function response turn comes immediately after a function call turn - - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 - """ - - contents: list[BedrockMessageBlock] = [] - msg_i = 0 - - messages = BedrockConverseMessagesProcessor._initial_message_setup( - messages, model, llm_provider, user_continue_message - ) - - while msg_i < len(messages): - user_content: list[BedrockContentBlock] = [] - init_msg_i = msg_i - ## MERGE CONSECUTIVE USER CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "user": - message_block = get_user_message_block_or_continue_message( - message=messages[msg_i], - user_continue_message=user_continue_message, - ) - if isinstance(message_block["content"], list): - _parts: list[BedrockContentBlock] = [] - for element in message_block["content"]: - if isinstance(element, dict): - if element["type"] == "text": - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "guarded_text": - # Wrap guarded_text in guardContent block - _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) - _parts.append(_part) - elif element["type"] in ("grounding_source", "query"): - # Contextual grounding tags are guardrail metadata; the - # model only needs the underlying text, so render them as - # plain text on the generate path. - _part = BedrockContentBlock(text=element["text"]) - _parts.append(_part) - elif element["type"] == "image_url": - format: str | None = None - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - format = element["image_url"].get("format") - else: - image_url = element["image_url"] - _part = BedrockImageProcessor.process_image_sync( - image_url=image_url, - format=format, - ) - _parts.append(_part) - elif element["type"] == "file": - _part = BedrockConverseMessagesProcessor._process_file_message( - message=cast(ChatCompletionFileObject, element) - ) - _parts.append(_part) - elif element["type"] == "document": - _part = BedrockConverseMessagesProcessor._process_document_message(element) - _parts.append(_part) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - if ( - _parts - and message_block.get("cache_control") is not None - and not any("cachePoint" in part for part in _parts) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - _parts.append(_cache_point_block) - user_content.extend(_parts) - elif message_block["content"] and isinstance(message_block["content"], str): - _part = BedrockContentBlock(text=messages[msg_i]["content"]) - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block, block_type="content_block", model=model - ) - user_content.append(_part) - if _cache_point_block is not None: - user_content.append(_cache_point_block) - - msg_i += 1 - if user_content: - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=user_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(user_content) - else: - contents.append(BedrockMessageBlock(role="user", content=user_content)) - - ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## - tool_content: list[BedrockContentBlock] = [] - while msg_i < len(messages) and messages[msg_i]["role"] == "tool": - tool_call_result = _convert_to_bedrock_tool_call_result(messages[msg_i]) - current_message = messages[msg_i] - - # Add the tool result first - tool_content.append(tool_call_result) - - # Check if we need to add a separate cachePoint block - tool_msg_cache_control = None - - # Check for message-level cache_control - if current_message.get("cache_control", None) is not None: - tool_msg_cache_control = current_message["cache_control"] - # Check for content-level cache_control in list content - elif isinstance(current_message.get("content"), list): - for content_element in current_message["content"]: - if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: - tool_msg_cache_control = content_element["cache_control"] - break - - # Add a separate cachePoint block if cache_control is present - if tool_msg_cache_control is not None: - cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - {"cache_control": tool_msg_cache_control}, - block_type="content_block", - model=model, - ) - if cache_point_block is not None: - tool_content.append(cache_point_block) - - msg_i += 1 - # Deduplicate toolResult blocks with the same toolUseId - tool_content = _deduplicate_bedrock_tool_content(tool_content) - if tool_content: - # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) - if len(contents) > 0 and contents[-1]["role"] == "user": - if assistant_continue_message is not None or litellm.modify_params is True: - # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) - contents = _insert_assistant_continue_message( - messages=contents, - assistant_continue_message=assistant_continue_message, - ) - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - else: - verbose_logger.warning( - "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." - ) - contents[-1]["content"].extend(tool_content) - else: - contents.append(BedrockMessageBlock(role="user", content=tool_content)) - assistant_content: list[BedrockContentBlock] = [] - ## MERGE CONSECUTIVE ASSISTANT CONTENT ## - while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": - assistant_message_block = get_assistant_message_block_or_continue_message( - message=messages[msg_i], - assistant_continue_message=assistant_continue_message, - ) - assistant_message_content_start: int = len(assistant_content) - _assistant_content = assistant_message_block.get("content", None) - thinking_blocks = cast( - list[ChatCompletionThinkingBlock] | None, - assistant_message_block.get("thinking_blocks"), - ) - - if thinking_blocks is not None: - converted_thinking_blocks = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks - ) - ) - assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=converted_thinking_blocks, - assistant_parts=assistant_content, - ) - - if _assistant_content is not None and isinstance(_assistant_content, list): - assistants_parts: list[BedrockContentBlock] = [] - for element in _assistant_content: - if isinstance(element, dict): - if element["type"] == "thinking": - thinking_block = ( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] - ) - ) - assistants_parts = ( - BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( - thinking_blocks=thinking_block, - assistant_parts=assistants_parts, - ) - ) - elif element["type"] == "text": - # AWS Bedrock doesn't allow empty or whitespace-only text content - # Skip completely empty strings to avoid blank content blocks - if element.get("text", "").strip(): - assistants_part = BedrockContentBlock(text=element["text"]) - assistants_parts.append(assistants_part) - elif element["type"] == "image_url": - if isinstance(element["image_url"], dict): - image_url = element["image_url"]["url"] - else: - image_url = element["image_url"] - assistants_part = BedrockImageProcessor.process_image_sync(image_url=image_url) - assistants_parts.append(assistants_part) - # Add cache point block for assistant content elements - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - message_block=cast(OpenAIMessageContentListBlock, element), - block_type="content_block", - model=model, - ) - if _cache_point_block is not None: - assistants_parts.append(_cache_point_block) - assistant_content.extend(assistants_parts) - elif _assistant_content is not None and isinstance(_assistant_content, str): - # Skip completely empty strings to avoid blank content blocks - if _assistant_content.strip(): - assistant_content.append(BedrockContentBlock(text=_assistant_content)) - # Add cache point block for assistant string content - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - _tool_calls = assistant_message_block.get("tool_calls", []) - if _tool_calls: - assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) - - if ( - isinstance(_assistant_content, list) - and len(assistant_content) > assistant_message_content_start - and assistant_message_block.get("cache_control") is not None - and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) - ): - _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) - if _cache_point_block is not None: - assistant_content.append(_cache_point_block) - - msg_i += 1 - - assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") - assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) - - if assistant_content: - contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) - - if msg_i == init_msg_i: # prevent infinite loops - raise litellm.BadRequestError( - message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", - model=model, - llm_provider=llm_provider, - ) - - return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) - - -def make_valid_bedrock_tool_name(input_tool_name: str) -> str: - """Normalize tool names to Bedrock pattern [a-zA-Z][a-zA-Z0-9_-]*.""" - - def replace_invalid(char): - if char.isalnum() or char in ("_", "-"): - return char - return "_" - - # If the string is empty, return a default valid identifier - if input_tool_name is None or len(input_tool_name) == 0: - return input_tool_name - bedrock_tool_name = copy.copy(input_tool_name) - # If it doesn't start with a letter, prepend 'a' - if not bedrock_tool_name[0].isalpha(): - bedrock_tool_name = "a" + bedrock_tool_name - - # Replace any invalid characters with underscores - valid_string: Final = "".join(replace_invalid(char) for char in bedrock_tool_name) - - if input_tool_name != valid_string: - # passed tool name was formatted to become valid - # store it internally so we can use for the response - litellm.bedrock_tool_name_mappings.set_cache(key=valid_string, value=input_tool_name) - - return valid_string - - -def add_cache_point_tool_block(tool: dict, model: str | None = None) -> BedrockToolBlock | None: - from litellm.llms.bedrock.common_utils import ( - bedrock_model_accepts_cache_points, - is_claude_4_5_on_bedrock, - ) - - cache_control: Final = tool.get("cache_control", None) - if cache_control is not None and bedrock_model_accepts_cache_points(model): - cache_point: Final = cache_control.get("type", "ephemeral") - if cache_point == "ephemeral": - cache_point_block: Final[CachePointBlock] = {"type": "default"} - if isinstance(cache_control, dict) and "ttl" in cache_control: - ttl: Final = cache_control["ttl"] - if ttl in ["5m", "1h"] and model is not None and is_claude_4_5_on_bedrock(model): - cache_point_block["ttl"] = ttl - return {"cachePoint": cache_point_block} - return None - - -def _is_bedrock_tool_block(tool: dict) -> bool: - """ - Check if a tool is already a BedrockToolBlock. - - BedrockToolBlock has one of: systemTool, toolSpec, or cachePoint. - This is used to detect tools that are already in Bedrock format - (e.g., systemTool for Nova grounding) vs OpenAI-style function tools - that need transformation. - - Args: - tool: The tool dict to check - - Returns: - True if the tool is already a BedrockToolBlock, False otherwise - - Examples: - >>> _is_bedrock_tool_block({"systemTool": {"name": "nova_grounding"}}) - True - >>> _is_bedrock_tool_block({"type": "function", "function": {...}}) - False - """ - return isinstance(tool, dict) and ("systemTool" in tool or "toolSpec" in tool or "cachePoint" in tool) - - -def _bedrock_tools_pt(tools: list, model: str | None = None) -> list[BedrockToolBlock]: - """ - OpenAI tools looks like: - tools = [ - { - "type": "function", - "function": { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, - }, - "required": ["location"], - }, - } - } - ] - """ - """ - Bedrock toolConfig looks like: - "tools": [ - { - "toolSpec": { - "name": "top_song", - "description": "Get the most popular song played on a radio station.", - "inputSchema": { - "json": { - "type": "object", - "properties": { - "sign": { - "type": "string", - "description": "The call sign for the radio station for which you want the most popular song. Example calls signs are WZPZ, and WKRP." - } - }, - "required": [ - "sign" - ] - } - } - } - } - ] - """ - from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs - from litellm.llms.bedrock.common_utils import ( - bedrock_converse_supports_strict_tools, - normalize_json_schema_custom_types_to_object, - ) - - _valid_json_schema_root_types = frozenset(("array", "boolean", "integer", "null", "number", "object", "string")) - # Only Claude on Bedrock honours strict tool schemas; other families - # (Nova, Llama, GPT-OSS) reject the strict field outright. Opus 4.7/4.8 - # also reject `strict` on Bedrock Converse (see #31582) — their validator - # maps toolSpec to the native Anthropic tool shape, which has no strict - # field, even though Anthropic's native API accepts it as a top-level key. - supports_strict_tools: Final = bool(model and bedrock_converse_supports_strict_tools(model)) - tool_block_list: Final[list[BedrockToolBlock]] = [] - for tool_idx, tool in enumerate(tools): - # Check if tool is already a BedrockToolBlock (e.g., systemTool for Nova grounding) - if _is_bedrock_tool_block(tool): - # Already a BedrockToolBlock, pass it through - tool_block_list.append(tool) - continue - - # Responses built-in tools (web_search, image_generation, namespace, tool_search, - # custom) carry neither an OpenAI "function" nor an Anthropic "input_schema" and have - # no Bedrock toolSpec equivalent; drop them instead of emitting an empty junk toolSpec. - if isinstance(tool, dict) and "function" not in tool and "input_schema" not in tool: - continue - - # OpenAI function tools, or Anthropic Messages / Claude Code ({name, input_schema, type, ...}) - if isinstance(tool, dict) and "input_schema" in tool and "function" not in tool: - parameters = copy.deepcopy(tool.get("input_schema") or {"type": "object", "properties": {}}) - raw_name = tool.get("name", "") or "" - _tool_description = tool.get("description", None) - else: - parameters = copy.deepcopy(tool.get("function", {}).get("parameters", {"type": "object", "properties": {}})) - raw_name = tool.get("function", {}).get("name", "") or "" - _tool_description = tool.get("function", {}).get("description", None) - - if not (raw_name and str(raw_name).strip()): - raw_name = f"litellm_unnamed_tool_{tool_idx}" - - # related issue: https://github.com/BerriAI/litellm/issues/5007 - # Bedrock tool names must satisfy pattern: [a-zA-Z][a-zA-Z0-9_-]* - name = make_valid_bedrock_tool_name(input_tool_name=raw_name) - if _tool_description: # bedrock doesn't accept empty "" or None descriptions - description = _tool_description - else: - description = name - - defs = parameters.pop("$defs", {}) - defs_copy = copy.deepcopy(defs) - # Expand $ref references in parameters using the definitions - # Note: We don't pre-flatten defs as that causes exponential memory growth - # with circular references (see issue #19098). unpack_defs handles nested - # refs recursively and correctly detects/skips circular references. - unpack_defs(parameters, defs_copy) - normalize_json_schema_custom_types_to_object(parameters) - if parameters.get("type") not in _valid_json_schema_root_types: - parameters["type"] = "object" - tool_block = cast( - BedrockToolBlock, - BedrockToolSpec( - name=name, - description=description, - parameters=parameters, - strict=tool.get("function", {}).get("strict", None), - supports_strict_tools=supports_strict_tools, - ), - ) - tool_block_list.append(tool_block) - - ## ADD CACHE POINT TOOL BLOCK ## - cache_point_tool_block = add_cache_point_tool_block(tool, model=model) - if cache_point_tool_block is not None: - tool_block_list.append(cache_point_tool_block) - - return tool_block_list - - -# Function call template -def function_call_prompt(messages: list, functions: list): - function_prompt = """Produce JSON OUTPUT ONLY! Adhere to this format {"name": "function_name", "arguments":{"argument_name": "argument_value"}} The following functions are available to you:""" - for function in functions: - function_prompt += f"""\n{function}\n""" - - function_added_to_prompt = False - for message in messages: - if "system" in message["role"]: - if isinstance(message["content"], str): - message["content"] += f""" {function_prompt}""" - else: - message["content"].append({"type": "text", "text": f""" {function_prompt}"""}) - function_added_to_prompt = True - - if function_added_to_prompt is False: - messages.append({"role": "system", "content": f"""{function_prompt}"""}) - - return messages - - -def response_schema_prompt(model: str, response_schema: dict) -> str: - """ - Decides if a user-defined custom prompt or default needs to be used - - Returns the prompt str that's passed to the model as a user message - """ - custom_prompt_details: dict | None = None - response_schema_as_message: Final = [{"role": "user", "content": f"{response_schema}"}] - if f"{model}/response_schema_prompt" in litellm.custom_prompt_dict: - custom_prompt_details = litellm.custom_prompt_dict[ - f"{model}/response_schema_prompt" - ] # allow user to define custom response schema prompt by model - elif "response_schema_prompt" in litellm.custom_prompt_dict: - custom_prompt_details = litellm.custom_prompt_dict["response_schema_prompt"] - - if custom_prompt_details is not None: - return custom_prompt( - role_dict=custom_prompt_details["roles"], - initial_prompt_value=custom_prompt_details["initial_prompt_value"], - final_prompt_value=custom_prompt_details["final_prompt_value"], - messages=response_schema_as_message, - ) - else: - return default_response_schema_prompt(response_schema=response_schema) - - -def default_response_schema_prompt(response_schema: dict) -> str: - """ - Used if provider/model doesn't support 'response_schema' param. - - This is the default prompt. Allow user to override this with a custom_prompt. - """ - prompt_str: Final = f"""Use this JSON schema: - ```json - {response_schema} - ```""" - return prompt_str - - -# Custom prompt template -def custom_prompt( - role_dict: dict, - messages: list, - initial_prompt_value: str = "", - final_prompt_value: str = "", - bos_token: str = "", - eos_token: str = "", -) -> str: - prompt = bos_token + initial_prompt_value - bos_open = True - ## a bos token is at the start of a system / human message - ## an eos token is at the end of the assistant response to the message - for message in messages: - role = message["role"] - - if role in ["system", "human"] and not bos_open: - prompt += bos_token - bos_open = True - - pre_message_str = ( - role_dict[role]["pre_message"] if role in role_dict and "pre_message" in role_dict[role] else "" - ) - post_message_str = ( - role_dict[role]["post_message"] if role in role_dict and "post_message" in role_dict[role] else "" - ) - if isinstance(message["content"], str): - prompt += pre_message_str + message["content"] + post_message_str - elif isinstance(message["content"], list): - text_str = "" - for content in message["content"]: - if content.get("text", None) is not None and isinstance(content["text"], str): - text_str += content["text"] - prompt += pre_message_str + text_str + post_message_str - - if role == "assistant": - prompt += eos_token - bos_open = False - - prompt += final_prompt_value - return prompt - - -def prompt_factory( - model: str, - messages: list, - custom_llm_provider: str | None = None, - api_key: str | None = None, -): - original_model_name: Final = model - model = model.lower() - if custom_llm_provider == "ollama": - return ollama_pt(model=model, messages=messages) - elif custom_llm_provider == "anthropic": - if litellm.AnthropicTextConfig._is_anthropic_text_model(model): - return anthropic_pt(messages=messages) - return anthropic_messages_pt(messages=messages, model=model, llm_provider=custom_llm_provider) - elif custom_llm_provider == "anthropic_xml": - return anthropic_messages_pt_xml(messages=messages) - elif custom_llm_provider == "gemini": - if ( - model == "gemini-pro-vision" - or litellm.supports_vision(model=model) - or litellm.supports_vision(model=custom_llm_provider + "/" + model) - ): - return _gemini_vision_convert_messages(messages=messages) - else: - return gemini_text_image_pt(messages=messages) - elif custom_llm_provider == "mistral": - return litellm.MistralConfig()._transform_messages(messages=messages, model=model) - elif custom_llm_provider == "bedrock": - if "amazon.titan-text" in model: - return amazon_titan_pt(messages=messages) - elif "anthropic." in model: - if any(_ in model for _ in ["claude-2.1", "claude-v2:1"]): - return claude_2_1_pt(messages=messages) - else: - return anthropic_pt(messages=messages) - elif "mistral." in model: - return mistral_instruct_pt(messages=messages) - elif "llama2" in model and "chat" in model: - return llama_2_chat_pt(messages=messages) - elif ("llama3" in model or "llama4" in model) and "instruct" in model: - return hf_chat_template( - model="meta-llama/Meta-Llama-3-8B-Instruct", - messages=messages, - ) - - elif custom_llm_provider == "clarifai": - if "claude" in model: - return anthropic_pt(messages=messages) - - elif custom_llm_provider == "perplexity": - for message in messages: - message.pop("name", None) - return messages - elif custom_llm_provider == "azure_text": - return azure_text_pt(messages=messages) - elif custom_llm_provider == "watsonx": - from litellm.llms.watsonx.chat.transformation import IBMWatsonXChatConfig - - return IBMWatsonXChatConfig.apply_prompt_template(model=model, messages=messages) - - try: - if "meta-llama/llama-2" in model and "chat" in model: - return llama_2_chat_pt(messages=messages) - elif ("meta-llama/llama-3" in model or "meta-llama-3" in model) and "instruct" in model: - return hf_chat_template( - model="meta-llama/Meta-Llama-3-8B-Instruct", - messages=messages, - ) - elif ( - "tiiuae/falcon" in model - ): # Note: for the instruct models, it's best to use a User: .., Assistant:.. approach in your prompt template. - if model == "tiiuae/falcon-180B-chat": - return falcon_chat_pt(messages=messages) - elif "instruct" in model: - return falcon_instruct_pt(messages=messages) - elif "mosaicml/mpt" in model: - if "chat" in model: - return mpt_chat_pt(messages=messages) - elif "codellama/codellama" in model or "togethercomputer/codellama" in model: - if "instruct" in model: - return llama_2_chat_pt( - messages=messages - ) # https://huggingface.co/blog/codellama#conversational-instructions - elif "wizardlm/wizardcoder" in model: - return wizardcoder_pt(messages=messages) - elif "phind/phind-codellama" in model: - return phind_codellama_pt(messages=messages) - elif "togethercomputer/llama-2" in model and ("instruct" in model or "chat" in model): - return llama_2_chat_pt(messages=messages) - elif model in [ - "gryphe/mythomax-l2-13b", - "gryphe/mythomix-l2-13b", - "gryphe/mythologic-l2-13b", - ]: - return alpaca_pt(messages=messages) - elif model.strip() == "meta-llama/llamaguard-7b": - chat_template = "{% if messages|length % 2 == 0 %}{% set role = 'Agent' %}{% else %}{% set role = 'User' %}{% endif %}{{ \"[INST] Task: Check if there is unsafe content in '\" + role + \"' messages in conversations according our safety policy with the below categories.\n\n\nO1: Violence and Hate. \nShould not\n- Help people plan or engage in violence.\n- Encourage people to engage in violence.\n- Express hateful, derogatory, or demeaning sentiments against people based on sensitive personal characteristics like their race, color, religion, national origin, sexual orientation, gender, gender identity, or disability.\n- Encourage discrimination against people based on those characteristics and should not use slurs or create insults based on those characteristics.\nCan\n- Provide information on violence and discrimination.\n- Discuss the topics of hate and violence, and can discuss historical events involving violence.\nO2: Sexual Content.\nShould not\n- Engage in sexually explicit (i.e., erotic) conversations and should not encourage people to engage in sexual activities.\nCan\n- Discuss topics related to sexuality and sexual education.\n- Provide general information regarding sexual activities.\n- Provide guidance on topics related to sexual health.\nO3: Criminal Planning.\nShould not\n- Help people plan or engage in criminal activities like theft, kidnapping, or financial crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in criminal activities or condone criminal behavior.\nCan\n- Help people understand how crimes are committed, provided that the response is not so detailed that it could be used to provide a roadmap for future crimes.\n- Help people understand how systems work that are used to prevent crimes, provided those descriptions do not provide detailed information on how to defeat those systems.\nO4: Guns and Illegal Weapons.\nShould not\n- Assist in the planning or execution of firearm or illegal weapons crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in firearm or illegal weapons crimes or condone such behavior.\nCan\n- Discuss firearms and the arguments for and against firearm ownership.\nO5: Regulated or Controlled Substances.\nShould not\n- Encourage or assist people in the trafficking, consumption, or use of regulated or controlled substances, including but not limited to illegal drugs, tobacco, alcohol, and cannabis.\n- Assist or encourage people to create such substances when it is illegal to do so.\nCan\n- Discuss the use of regulated or controlled substances, and can provide general information on those substances (ex: what their chemical composition is and what their effects are).\n- Provide information on how those substances are manufactured, provided that the information is not so detailed as to provide step-by-step instructions for their creation.\nO6: Self-Harm.\nShould not\n- Encourage people to harm themselves.\n- Romanticize or condone self-harm or suicide.\n- Provide information regarding the methods of suicide or self-harm.\n- Provide any information or advice, other than information regarding appropriate health resources, to people who express or imply that they wish to harm themselves.\nCan\n- Provide general information regarding self-harm and suicide (ex: statistics, the mental health challenges faced by those who take these actions).\nShould\n- Provide information regarding appropriate health resources to people who imply or express the desire to harm themselves.\n\n\n\n\n\" }}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{% set role = 'User' %}{% elif message['role'] == 'assistant' %}{% set role = 'Agent' %}{% endif %}{{ role + ': ' + content.strip() + '\n\n' }}{% endfor %}{{ \"\n\nProvide your safety assessment for \" + role + \" in the above conversation:\n- First line must read 'safe' or 'unsafe'.\n- If unsafe, a second line must include a comma-separated list of violated categories. [/INST]\" }}" - return hf_chat_template(model=model, messages=messages, chat_template=chat_template) - else: - return hf_chat_template(original_model_name, messages) - except Exception: - return default_pt( - messages=messages - ) # default that covers Bloom, T-5, any non-chat tuned model (e.g. base Llama2) - - -def get_attribute_or_key(tool_or_function, attribute, default=None): - if hasattr(tool_or_function, attribute): - return getattr(tool_or_function, attribute) - if isinstance(tool_or_function, Mapping): - return tool_or_function.get(attribute, default) - return default - - -class NormalizedToolCall(TypedDict): - id: str | None - name: str | None - arguments: dict[str, object] - - -_ArgumentObjects: TypeAlias = tuple[dict[str, object], ...] -_ParsedToolCall: TypeAlias = tuple[str | None, str | None, _ArgumentObjects] - - -def _optional_call_id(value: object) -> str | None: - if isinstance(value, str) and value: - return value - return None - - -def _optional_tool_name(value: object) -> str | None: - if isinstance(value, str): - return value - return None - - -def _split_tool_call_ids(calls: Sequence[tuple[str | None, int]]) -> tuple[tuple[str | None, ...], ...]: - taken: Final = frozenset(_sanitize_anthropic_tool_use_id(call_id) for call_id, _ in calls if call_id) - - def fresh(call_id: str) -> Iterator[str]: - return filter( - lambda candidate: _sanitize_anthropic_tool_use_id(candidate) not in taken, - (f"{call_id}__concat_{n}" for n in itertools.count(1)), - ) - - suffixes: Final = MappingProxyType( - {_sanitize_anthropic_tool_use_id(call_id): fresh(call_id) for call_id, count in calls if call_id and count > 1} - ) - return tuple( - ( - call_id, - *(next(suffixes[_sanitize_anthropic_tool_use_id(call_id)]) for _ in range(count - 1)), - ) - if call_id - else (None,) * count - for call_id, count in calls - ) - - -def _parse_tool_call_arguments(raw: object, tool_name: str | None, context: str) -> _ArgumentObjects: - # Anthropic's tool_use blocks already carry a parsed dict in "input"; - # chat completions and the Responses API carry a JSON string that may be - # truncated by the model, so route those through the repair-aware parser. - if isinstance(raw, dict): - return (raw,) - if not isinstance(raw, str): - return ({},) - normalized_raw: Final = "{}" if raw == REDACTED_BY_LITELLM else raw - try: - parsed: Final = parse_tool_call_arguments(normalized_raw, tool_name=tool_name, context=context) - except ValueError as e: - salvaged: Final = salvage_concatenated_tool_arguments(normalized_raw) - if salvaged: - verbose_logger.warning( - "Recovered %d tool call(s) from concatenated JSON arguments for tool '%s' (%s)", - len(salvaged), - tool_name or "", - context, - ) - return salvaged - verbose_logger.warning("Failed to parse tool call arguments: %s", e) - return ({},) - return (parsed,) if isinstance(parsed, dict) else ({},) - - -def _choice_tool_calls(choice: object) -> tuple[object, ...]: - message: Final = get_attribute_or_key(choice, "message", None) - tool_calls: Final = get_attribute_or_key(message, "tool_calls", None) if message is not None else None - if isinstance(tool_calls, list): - return tuple(tool_calls) - return () - - -def _selected_choices(response: object, include_all_choices: bool) -> tuple[object, ...]: - choices: Final = get_attribute_or_key(response, "choices", None) - if not isinstance(choices, list) or not choices: - return () - if include_all_choices: - return tuple(choices) - return (choices[0],) - - -def _parsed_chat_tool_call(tool_call: object) -> _ParsedToolCall | None: - function: Final = get_attribute_or_key(tool_call, "function", None) - if function is None: - return None - name: Final = _optional_tool_name(get_attribute_or_key(function, "name")) - return ( - _optional_call_id(get_attribute_or_key(tool_call, "id")), - name, - _parse_tool_call_arguments( - get_attribute_or_key(function, "arguments", "{}"), - tool_name=name, - context="chat completions", - ), - ) - - -def _parsed_calls_in_choice(choice: object) -> tuple[_ParsedToolCall, ...]: - return tuple( - parsed for tool_call in _choice_tool_calls(choice) if (parsed := _parsed_chat_tool_call(tool_call)) is not None - ) - - -def _parsed_chat_tool_calls(response: object, include_all_choices: bool) -> tuple[_ParsedToolCall, ...]: - grouped: Final = tuple( - _parsed_calls_in_choice(choice) for choice in _selected_choices(response, include_all_choices) - ) - return tuple(itertools.chain.from_iterable(grouped)) - - -def _normalized_tool_calls_for_parse( - name: str | None, - call_ids: tuple[str | None, ...], - arguments: _ArgumentObjects, -) -> tuple[NormalizedToolCall, ...]: - return tuple( - NormalizedToolCall(id=call_id, name=name, arguments=argument) - for call_id, argument in zip(call_ids, arguments, strict=True) - ) - - -def _normalized_tool_calls_from_parses(parses: Sequence[_ParsedToolCall]) -> tuple[NormalizedToolCall, ...]: - id_groups: Final = _split_tool_call_ids(tuple((call_id, len(arguments)) for call_id, _, arguments in parses)) - grouped: Final = tuple( - _normalized_tool_calls_for_parse(name, call_ids, arguments) - for (_, name, arguments), call_ids in zip(parses, id_groups, strict=True) - ) - return tuple(itertools.chain.from_iterable(grouped)) - - -def _tool_calls_from_chat_completion_response( - response: object, include_all_choices: bool = False -) -> tuple[NormalizedToolCall, ...]: - return _normalized_tool_calls_from_parses(_parsed_chat_tool_calls(response, include_all_choices)) - - -def _response_function_calls(response: object) -> tuple[object, ...]: - output: Final = get_attribute_or_key(response, "output", None) - if not isinstance(output, list): - return () - return tuple(item for item in output if get_attribute_or_key(item, "type") == "function_call") - - -def _parsed_response_tool_call(item: object) -> _ParsedToolCall: - name: Final = _optional_tool_name(get_attribute_or_key(item, "name")) - raw_id: Final = get_attribute_or_key(item, "call_id") or get_attribute_or_key(item, "id") - return ( - _optional_call_id(raw_id), - name, - _parse_tool_call_arguments( - get_attribute_or_key(item, "arguments", "{}"), - tool_name=name, - context="responses API", - ), - ) - - -def _tool_calls_from_responses_api_response(response: object) -> tuple[NormalizedToolCall, ...]: - parses: Final = tuple(_parsed_response_tool_call(item) for item in _response_function_calls(response)) - return _normalized_tool_calls_from_parses(parses) - - -def _tool_calls_from_anthropic_messages_response(response: object) -> list[NormalizedToolCall]: - content: Final = get_attribute_or_key(response, "content", None) - if not isinstance(content, list): - return [] - result: Final[list[NormalizedToolCall]] = [] - for block in content: - if get_attribute_or_key(block, "type") != "tool_use": - continue - raw_input = get_attribute_or_key(block, "input", {}) - result.append( - NormalizedToolCall( - id=get_attribute_or_key(block, "id"), - name=get_attribute_or_key(block, "name"), - arguments=raw_input if isinstance(raw_input, dict) else {}, - ) - ) - return result - - -def get_tool_calls_from_response(response: object, include_all_choices: bool = False) -> list[NormalizedToolCall]: - """ - Extract tool/function calls from a response object into a normalized - ``{"id", "name", "arguments"}`` shape, regardless of which API surface - produced it: chat completions (``choices[].message.tool_calls``), - the Responses API (``output`` items of type ``function_call``), or the - Anthropic Messages API (``content`` blocks of type ``tool_use``). - - ``include_all_choices`` decides the chat-completions scope: the default - reads only ``choices[0]``, which is what consumers that act on THE reply - (e.g. guardrails rebuilding the primary assistant message) want; usage - accounting passes True because every choice of an ``n>1`` request costs - money and its tool calls really ran. The other surfaces have a single - output, so the flag has no effect on them. - - Callers that only care about a specific tool should filter the result by - ``name`` themselves -- this returns every tool call found. - """ - chat_tool_calls: Final = _tool_calls_from_chat_completion_response( - response, include_all_choices=include_all_choices - ) - if chat_tool_calls: - return list(chat_tool_calls) - for extractor in ( - _tool_calls_from_responses_api_response, - _tool_calls_from_anthropic_messages_response, - ): - tool_calls = extractor(response) - if tool_calls: - return list(tool_calls) - return [] - - -def has_tool_with_name(tools: object, tool_name: str) -> bool: - """ - Check whether a tools list (as sent to an LLM) includes a tool with the - given name, regardless of shape: OpenAI-style function tools - (``{"type": "function", "function": {"name": ...}}``) or Anthropic's - native tool shape (a top-level ``"name"``, e.g. - ``{"name": ..., "input_schema": ...}``). Anthropic's documented client - tool format doesn't require a ``"type"`` key at all -- ``"custom"`` is - only one of several possible values -- so any non-OpenAI-shaped tool is - matched on its top-level ``"name"``. - """ - if not isinstance(tools, list): - return False - for tool in tools: - if not isinstance(tool, dict): - continue - function = tool.get("function") - if tool.get("type") == "function" and isinstance(function, dict): - if function.get("name") == tool_name: - return True - elif tool.get("name") == tool_name: - return True - return False - - -def resolve_structured_messages( - messages: list[dict[str, object]] | None, - request_kwargs: dict[str, Any], -) -> list[dict[str, object]] | None: - """ - Normalize a request's messages to OpenAI-spec chat-completions shape, - regardless of which API surface produced them (chat completions, - Anthropic /v1/messages, Responses API ``input``, etc). - - Returns ``messages`` unchanged if already present. Otherwise dispatches - through the guardrail translation handlers (the same per-surface - conversion logic guardrails use) to convert e.g. Responses API ``input`` - into a message list. Returns ``None`` if no messages could be resolved. - """ - if messages: - return messages - - from litellm.litellm_core_utils.api_route_to_call_types import ( - get_call_types_for_route, - ) - from litellm.llms import load_guardrail_translation_mappings - from litellm.types.utils import CallTypes - - mappings: Final = load_guardrail_translation_mappings() - call_type: CallTypes | None = None - - # 1. Try route-based inference from proxy metadata - route: Final = request_kwargs.get("litellm_metadata", {}).get("user_api_key_request_route") - if route: - call_types_list: Final = get_call_types_for_route(route) - if call_types_list: - for ct in call_types_list: - if ct in mappings: - call_type = ct - break - - # 2. Fallback: try each mapped handler until one produces messages - handlers_to_try: Final[list[Any]] = [] - if call_type is not None and call_type in mappings: - handlers_to_try.append(mappings[call_type]()) - else: - handlers_to_try.extend(handler_cls() for handler_cls in mappings.values()) - - for handler in handlers_to_try: - structured = handler.get_structured_messages(request_kwargs) - if structured: - return [msg if isinstance(msg, dict) else msg.model_dump() for msg in structured] - return None +import base64 +import copy +import hashlib +import itertools +import json +import mimetypes +import re +import xml.etree.ElementTree as ET +from bisect import bisect_left +from collections.abc import Iterator, Mapping, Sequence +from enum import Enum +from types import MappingProxyType +from typing import Any, Final, TypeAlias, TypedDict, cast, overload + +from jinja2.sandbox import ImmutableSandboxedEnvironment + +import litellm +import litellm.types +import litellm.types.llms +from litellm import verbose_logger +from litellm._uuid import uuid +from litellm.constants import REDACTED_BY_LITELLM +from litellm.litellm_core_utils.prompt_templates.mid_conversation_system import anthropic_system_messages +from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get +from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client +from litellm.types.files import get_file_extension_from_mime_type +from litellm.types.llms.anthropic import * +from litellm.types.llms.bedrock import CachePointBlock +from litellm.types.llms.bedrock import MessageBlock as BedrockMessageBlock +from litellm.types.llms.custom_http import httpxSpecialProvider +from litellm.types.llms.ollama import OllamaVisionModelObject +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionAssistantMessage, + ChatCompletionAssistantToolCall, + ChatCompletionFileObject, + ChatCompletionFileObjectFile, + ChatCompletionFunctionMessage, + ChatCompletionImageObject, + ChatCompletionImageUrlObject, + ChatCompletionTextObject, + ChatCompletionToolCallFunctionChunk, + ChatCompletionToolMessage, + ChatCompletionUserMessage, + OpenAIMessageContentListBlock, +) +from litellm.types.llms.vertex_ai import FunctionCall as VertexFunctionCall +from litellm.types.llms.vertex_ai import FunctionResponse as VertexFunctionResponse +from litellm.types.llms.vertex_ai import PartType as VertexPartType +from litellm.types.utils import GenericImageParsingChunk + +from .common_utils import ( + convert_content_list_to_str, + infer_content_type_from_url_and_content, + is_non_content_values_set, + is_unsignable_thinking_block, + parse_tool_call_arguments, + salvage_concatenated_tool_arguments, +) +from .image_handling import convert_url_to_base64 + + +def default_pt(messages): + return " ".join(message["content"] for message in messages) + + +def prompt_injection_detection_default_pt(): + return """Detect if a prompt is safe to run. Return 'UNSAFE' if not.""" + + +BAD_MESSAGE_ERROR_STR: Final = "Invalid Message " + +# Separator used to embed Gemini thought signatures in tool call IDs +# See: https://ai.google.dev/gemini-api/docs/thought-signatures +THOUGHT_SIGNATURE_SEPARATOR: Final = "__thought__" + +# used to interweave user messages, to ensure user/assistant alternating +DEFAULT_USER_CONTINUE_MESSAGE: Final = { + "role": "user", + "content": "Please continue.", +} # similar to autogen. Only used if `litellm.modify_params=True`. + +DEFAULT_USER_CONTINUE_MESSAGE_TYPED: Final = ChatCompletionUserMessage( + role="user", + content="Please continue.", +) + +# used to interweave assistant messages, to ensure user/assistant alternating +DEFAULT_ASSISTANT_CONTINUE_MESSAGE: Final = ChatCompletionAssistantMessage( + role="assistant", + content=[ + { + "type": "text", + "text": "Please continue.", + } + ], +) # similar to autogen. Only used if `litellm.modify_params=True`. + + +def map_system_message_pt(messages: list) -> list: + """ + Convert 'system' message to 'user' message if provider doesn't support 'system' role. + + Enabled via `completion(...,supports_system_message=False)` + + If next message is a user message or assistant message -> merge system prompt into it + + if next message is system -> append a user message instead of the system message + """ + + new_messages: Final = [] + for i, m in enumerate(messages): + if m["role"] == "system": + if i < len(messages) - 1: # Not the last message + next_m = messages[i + 1] + next_role = next_m["role"] + if next_role == "user" or next_role == "assistant": # Next message is a user or assistant message + # Merge system prompt into the next message + next_m["content"] = m["content"] + " " + next_m["content"] + elif next_role == "system": # Next message is a system message + # Append a user message instead of the system message + new_message = {"role": "user", "content": m["content"]} + new_messages.append(new_message) + else: # Last message + new_message = {"role": "user", "content": m["content"]} + new_messages.append(new_message) + else: # Not a system message + new_messages.append(m) + + return new_messages + + +# alpaca prompt template - for models like mythomax, etc. +def alpaca_pt(messages): + prompt: Final = custom_prompt( + role_dict={ + "system": { + "pre_message": "### Instruction:\n", + "post_message": "\n\n", + }, + "user": { + "pre_message": "### Instruction:\n", + "post_message": "\n\n", + }, + "assistant": {"pre_message": "### Response:\n", "post_message": "\n\n"}, + }, + bos_token="", + eos_token="", + messages=messages, + ) + return prompt + + +# Llama2 prompt template +def llama_2_chat_pt(messages): + prompt: Final = custom_prompt( + role_dict={ + "system": { + "pre_message": "[INST] <>\n", + "post_message": "\n<>\n [/INST]\n", + }, + "user": { # follow this format https://github.com/facebookresearch/llama/blob/77062717054710e352a99add63d160274ce670c6/llama/generation.py#L348 + "pre_message": "[INST] ", + "post_message": " [/INST]\n", + }, + "assistant": { + "post_message": "\n" # follows this - https://replicate.com/blog/how-to-prompt-llama + }, + }, + messages=messages, + bos_token="", + eos_token="", + ) + return prompt + + +def convert_to_ollama_image(openai_image_url: str): + try: + if openai_image_url.startswith("http"): + openai_image_url = convert_url_to_base64(url=openai_image_url) + + if openai_image_url.startswith("data:image/"): + # Extract the base64 image data + base64_data = openai_image_url.split("data:image/")[1].split(";base64,")[1] + else: + base64_data = openai_image_url + + return base64_data + except Exception as e: + if "Error: Unable to fetch image from URL" in str(e): + raise e + raise Exception( + """Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{base64_image}". """ + ) + + +def _handle_ollama_system_message(messages: list, prompt: str, msg_i: int) -> tuple[str, int]: + system_content_str = "" + ## MERGE CONSECUTIVE SYSTEM CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + msg_content = convert_content_list_to_str(messages[msg_i]) + system_content_str += msg_content + + msg_i += 1 + + return system_content_str, msg_i + + +def ollama_pt( + model: str, messages: list +) -> ( + str | OllamaVisionModelObject +): # https://github.com/ollama/ollama/blob/af4cf55884ac54b9e637cd71dadfe9b7a5685877/docs/modelfile.md#template + user_message_types: Final = {"user", "tool", "function"} + msg_i = 0 + images: Final = [] + prompt = "" + while msg_i < len(messages): + init_msg_i = msg_i + user_content_str = "" + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + msg_content = messages[msg_i].get("content") + if msg_content: + if isinstance(msg_content, list): + for m in msg_content: + if m.get("type", "") == "image_url": + if isinstance(m["image_url"], str): + images.append(m["image_url"]) + elif isinstance(m["image_url"], dict): + images.append(m["image_url"]["url"]) + elif m.get("type", "") == "text": + user_content_str += m["text"] + else: + # Tool message content will always be a string + user_content_str += msg_content + + msg_i += 1 + + if user_content_str: + prompt += f"### User:\n{user_content_str}\n\n" + + system_content_str, msg_i = _handle_ollama_system_message(messages, prompt, msg_i) + if system_content_str: + prompt += f"### System:\n{system_content_str}\n\n" + + assistant_content_str = "" + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_content_str += convert_content_list_to_str(messages[msg_i]) + + tool_calls = messages[msg_i].get("tool_calls") + ollama_tool_calls = [] + if tool_calls: + for call in tool_calls: + call_id: str = call["id"] + function_name: str = call["function"]["name"] + arguments = json.loads(call["function"]["arguments"]) + + ollama_tool_calls.append( + { + "id": call_id, + "type": "function", + "function": { + "name": function_name, + "arguments": arguments, + }, + } + ) + + if ollama_tool_calls: + assistant_content_str += f"Tool Calls: {json.dumps(ollama_tool_calls, indent=2)}" + + msg_i += 1 + + if assistant_content_str: + prompt += f"### Assistant:\n{assistant_content_str}\n\n" + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider="ollama", + ) + + response_dict: Final[OllamaVisionModelObject] = { + "prompt": prompt, + "images": images, + } + + return response_dict + + +def mistral_instruct_pt(messages): + # Following the Mistral example's https://huggingface.co/docs/transformers/main/chat_templating + prompt: Final = custom_prompt( + initial_prompt_value="", + role_dict={ + "system": { + "pre_message": "[INST] \n", + "post_message": " [/INST]\n", + }, + "user": {"pre_message": "[INST] ", "post_message": " [/INST]\n"}, + "assistant": {"pre_message": " ", "post_message": " "}, + }, + final_prompt_value="", + messages=messages, + ) + return prompt + + +# Falcon prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 +def falcon_instruct_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += message["content"] + else: + prompt += message["role"] + ":" + message["content"].replace("\r\n", "\n").replace("\n\n", "\n") + prompt += "\n\n" + + return prompt + + +def falcon_chat_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "System: " + message["content"] + elif message["role"] == "assistant": + prompt += "Falcon: " + message["content"] + elif message["role"] == "user": + prompt += "User: " + message["content"] + + return prompt + + +# MPT prompt template - from https://github.com/lm-sys/FastChat/blob/main/fastchat/conversation.py#L110 +def mpt_chat_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "<|im_start|>system" + message["content"] + "<|im_end|>" + "\n" + elif message["role"] == "assistant": + prompt += "<|im_start|>assistant" + message["content"] + "<|im_end|>" + "\n" + elif message["role"] == "user": + prompt += "<|im_start|>user" + message["content"] + "<|im_end|>" + "\n" + return prompt + + +# WizardCoder prompt template - https://huggingface.co/WizardLM/WizardCoder-Python-34B-V1.0#prompt-format +def wizardcoder_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += message["content"] + "\n\n" + elif message["role"] == "user": # map to 'Instruction' + prompt += "### Instruction:\n" + message["content"] + "\n\n" + elif message["role"] == "assistant": # map to 'Response' + prompt += "### Response:\n" + message["content"] + "\n\n" + return prompt + + +# Phind-CodeLlama prompt template - https://huggingface.co/Phind/Phind-CodeLlama-34B-v2#how-to-prompt-the-model +def phind_codellama_pt(messages): + prompt = "" + for message in messages: + if message["role"] == "system": + prompt += "### System Prompt\n" + message["content"] + "\n\n" + elif message["role"] == "user": + prompt += "### User Message\n" + message["content"] + "\n\n" + elif message["role"] == "assistant": + prompt += "### Assistant\n" + message["content"] + "\n\n" + return prompt + + +def _render_chat_template(env, chat_template: str, bos_token: str, eos_token: str, messages: list) -> str: + """ + Shared template rendering logic for both sync and async hf_chat_template + + Args: + env: Jinja2 environment + chat_template: Chat template string + bos_token: Beginning of sequence token + eos_token: End of sequence token + messages: Messages to render + + Returns: + Rendered template string + """ + try: + template: Final = env.from_string(chat_template) + except Exception as e: + raise e + + def _is_system_in_template(): + try: + # Try rendering the template with a system message + template.render( + messages=[{"role": "system", "content": "test"}], + eos_token="", + bos_token="", + ) + return True + # This will be raised if Jinja attempts to render the system message and it can't + except Exception: + return False + + try: + rendered_text = "" + # Render the template with the provided values + if _is_system_in_template(): + rendered_text = template.render( + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + add_generation_prompt=True, + ) + else: + # treat a system message as a user message, if system not in template + reformatted_messages: Final = [] + try: + for message in messages: + if message["role"] == "system": + reformatted_messages.append({"role": "user", "content": message["content"]}) + else: + reformatted_messages.append(message) + rendered_text = template.render( + bos_token=bos_token, + eos_token=eos_token, + messages=reformatted_messages, + add_generation_prompt=True, + ) + except Exception as e: + if "Conversation roles must alternate user/assistant" in str(e): + # reformat messages to ensure user/assistant are alternating + new_messages: Final = [] + for i in range(len(reformatted_messages) - 1): + new_messages.append(reformatted_messages[i]) + if reformatted_messages[i]["role"] == reformatted_messages[i + 1]["role"]: + if reformatted_messages[i]["role"] == "user": + new_messages.append({"role": "assistant", "content": ""}) + else: + new_messages.append({"role": "user", "content": ""}) + new_messages.append(reformatted_messages[-1]) + rendered_text = template.render(bos_token=bos_token, eos_token=eos_token, messages=new_messages) + + return rendered_text + except Exception as e: + raise Exception(f"Error rendering template - {e}") # don't use verbose_logger.exception, if exception is raised + + +async def _afetch_and_extract_template( + model: str, chat_template: str | None, get_config_fn, get_template_fn +) -> tuple[str, str, str]: + """ + Async version: Fetch template and tokens from HuggingFace. + + Returns: (chat_template, bos_token, eos_token) + """ + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _extract_token_value, + ) + + bos_token = "" + eos_token = "" + + if chat_template is None: + # Fetch or retrieve cached tokenizer config + if model in litellm.known_tokenizer_config: + tokenizer_config = litellm.known_tokenizer_config[model] + else: + tokenizer_config = await get_config_fn(hf_model_name=model) + litellm.known_tokenizer_config.update({model: tokenizer_config}) + + # Try to get chat template from tokenizer_config.json first + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + and "chat_template" in tokenizer_config["tokenizer"] + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + chat_template = tokenizer_data["chat_template"] + else: + # Fallback: Try to fetch chat template from separate .jinja file + template_result: Final = await get_template_fn(hf_model_name=model) + if template_result.get("status") == "success": + chat_template = template_result["chat_template"] + # Still try to get tokens from tokenizer_config if available + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + else: + raise Exception("No chat template found") + + return chat_template, bos_token, eos_token + + +def _fetch_and_extract_template( + model: str, chat_template: str | None, get_config_fn, get_template_fn +) -> tuple[str, str, str]: + """ + Sync version: Fetch template and tokens from HuggingFace. + + Returns: (chat_template, bos_token, eos_token) + """ + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _extract_token_value, + ) + + bos_token = "" + eos_token = "" + + if chat_template is None: + # Fetch or retrieve cached tokenizer config + if model in litellm.known_tokenizer_config: + tokenizer_config = litellm.known_tokenizer_config[model] + else: + tokenizer_config = get_config_fn(hf_model_name=model) + litellm.known_tokenizer_config.update({model: tokenizer_config}) + + # Try to get chat template from tokenizer_config.json first + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + and "chat_template" in tokenizer_config["tokenizer"] + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + chat_template = tokenizer_data["chat_template"] + else: + # Fallback: Try to fetch chat template from separate .jinja file + template_result: Final = get_template_fn(hf_model_name=model) + if template_result.get("status") == "success": + chat_template = template_result["chat_template"] + # Still try to get tokens from tokenizer_config if available + if ( + tokenizer_config.get("status") == "success" + and "tokenizer" in tokenizer_config + and isinstance(tokenizer_config["tokenizer"], dict) + ): + tokenizer_data: dict = tokenizer_config["tokenizer"] + bos_token = _extract_token_value(token_value=tokenizer_data.get("bos_token")) + eos_token = _extract_token_value(token_value=tokenizer_data.get("eos_token")) + else: + raise Exception("No chat template found") + + return chat_template, bos_token, eos_token + + +async def ahf_chat_template(model: str, messages: list, chat_template: str | None = None): + """HuggingFace chat template (async version)""" + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _aget_chat_template_file, + _aget_tokenizer_config, + strftime_now, + ) + + env: Final = ImmutableSandboxedEnvironment() + env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") + env.globals["strftime_now"] = strftime_now + + template, bos_token, eos_token = await _afetch_and_extract_template( + model=model, + chat_template=chat_template, + get_config_fn=_aget_tokenizer_config, + get_template_fn=_aget_chat_template_file, + ) + return _render_chat_template( + env=env, + chat_template=template, + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + ) + + +def hf_chat_template(model: str, messages: list, chat_template: str | None = None): + """HuggingFace chat template (sync version)""" + from litellm.litellm_core_utils.prompt_templates.huggingface_template_handler import ( + _get_chat_template_file, + _get_tokenizer_config, + strftime_now, + ) + + env: Final = ImmutableSandboxedEnvironment() + env.globals["raise_exception"] = lambda msg: Exception(f"Error message - {msg}") + env.globals["strftime_now"] = strftime_now + + template, bos_token, eos_token = _fetch_and_extract_template( + model=model, + chat_template=chat_template, + get_config_fn=_get_tokenizer_config, + get_template_fn=_get_chat_template_file, + ) + return _render_chat_template( + env=env, + chat_template=template, + bos_token=bos_token, + eos_token=eos_token, + messages=messages, + ) + + +def deepseek_r1_pt(messages): + return hf_chat_template(model="deepseek-r1/deepseek-r1-7b-instruct", messages=messages) + + +# Anthropic template +def claude_2_1_pt( + messages: list, +): # format - https://docs.anthropic.com/claude/docs/how-to-use-system-prompts + """ + Claude v2.1 allows system prompts (no Human: needed), but requires it be followed by Human: + - you can't just pass a system message + - you can't pass a system message and follow that with an assistant message + if system message is passed in, you can only do system, human, assistant or system, human + + if a system message is passed in and followed by an assistant message, insert a blank human message between them. + + Additionally, you can "put words in Claude's mouth" by ending with an assistant message. + See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth + """ + + class AnthropicConstants(Enum): + HUMAN_PROMPT = "\n\nHuman: " + AI_PROMPT = "\n\nAssistant: " + + prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "user": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{message['content']}" + elif message["role"] == "assistant": + if idx > 0 and messages[idx - 1]["role"] == "system": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}" # Insert a blank human message + prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" + if messages[-1]["role"] != "assistant": + prompt += f"{AnthropicConstants.AI_PROMPT.value}" # prompt must end with \"\n\nAssistant: " turn + return prompt + + +### IBM Granite + + +def ibm_granite_pt(messages: list): + """ + IBM's Granite models uses the template: + <|system|> {system_message} <|user|> {user_message} <|assistant|> {assistant_message} + + See: https://www.ibm.com/docs/en/watsonx-as-a-service?topic=solutions-supported-foundation-models + """ + return custom_prompt( + messages=messages, + role_dict={ + "system": { + "pre_message": "<|system|>\n", + "post_message": "\n", + }, + "user": { + "pre_message": "<|user|>\n", + # Assistant tag is needed in the prompt after the user message + # to avoid the model completing the users sentence before it answers + # https://www.ibm.com/docs/en/watsonx/w-and-w/2.0.x?topic=models-granite-13b-chat-v2-prompting-tips#chat + "post_message": "\n<|assistant|>\n", + }, + "assistant": { + "pre_message": "", + "post_message": "\n", + }, + }, + ).strip() + + +### ANTHROPIC ### + + +def anthropic_pt( + messages: list, +): # format - https://docs.anthropic.com/claude/reference/complete_post + """ + You can "put words in Claude's mouth" by ending with an assistant message. + See: https://docs.anthropic.com/claude/docs/put-words-in-claudes-mouth + """ + + class AnthropicConstants(Enum): + HUMAN_PROMPT = "\n\nHuman: " + AI_PROMPT = "\n\nAssistant: " + + prompt = "" + for idx, message in enumerate(messages): # needs to start with `\n\nHuman: ` and end with `\n\nAssistant: ` + if message["role"] == "user": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{AnthropicConstants.HUMAN_PROMPT.value}{message['content']}" + else: + prompt += f"{AnthropicConstants.AI_PROMPT.value}{message['content']}" + if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` + prompt = f"{AnthropicConstants.HUMAN_PROMPT.value}" + prompt + if messages[-1]["role"] != "assistant": + prompt += f"{AnthropicConstants.AI_PROMPT.value}" + return prompt + + +def construct_format_parameters_prompt(parameters: dict): + parameter_str = "\n" + for k, v in parameters.items(): + parameter_str += f"<{k}>" + parameter_str += f"{v}" + parameter_str += f"" + parameter_str += "\n" + return parameter_str + + +def construct_format_tool_for_claude_prompt(name, description, parameters): + constructed_prompt: Final = ( + "\n" + f"{name}\n" + "\n" + f"{description}\n" + "\n" + "\n" + f"{construct_format_parameters_prompt(parameters)}\n" + "\n" + "" + ) + return constructed_prompt + + +def construct_tool_use_system_prompt( + tools, +): # from https://github.com/anthropics/anthropic-cookbook/blob/main/function_calling/function_calling.ipynb + tool_str_list: Final = [] + for tool in tools: + tool_function = get_attribute_or_key(tool, "function") + tool_str = construct_format_tool_for_claude_prompt( + get_attribute_or_key(tool_function, "name"), + get_attribute_or_key(tool_function, "description", ""), + get_attribute_or_key(tool_function, "parameters", {}), + ) + tool_str_list.append(tool_str) + tool_use_system_prompt: Final = ( + "In this environment you have access to a set of tools you can use to answer the user's question.\n" + "\n" + "You may call them like this:\n" + "\n" + "\n" + "$TOOL_NAME\n" + "\n" + "<$PARAMETER_NAME>$PARAMETER_VALUE\n" + "...\n" + "\n" + "\n" + "\n" + "\n" + "Here are the tools available:\n" + "\n" + "\n".join([tool_str for tool_str in tool_str_list]) + "\n" + ) + return tool_use_system_prompt + + +def convert_generic_image_chunk_to_openai_image_obj( + image_chunk: GenericImageParsingChunk, +) -> str: + """ + Convert a generic image chunk to an OpenAI image object. + + Input: + GenericImageParsingChunk( + type="base64", + media_type="image/jpeg", + data="...", + ) + + Return: + "data:image/jpeg;base64,{base64_image}" + """ + media_type: Final = image_chunk["media_type"] + return "data:{};{},{}".format(media_type, image_chunk["type"], image_chunk["data"]) + + +def convert_to_anthropic_image_obj(openai_image_url: str, format: str | None) -> GenericImageParsingChunk: + """ + Input: + "image_url": "data:image/jpeg;base64,{base64_image}", + + Return: + "source": { + "type": "base64", + "media_type": "image/jpeg", + "data": {base64_image}, + } + """ + try: + if openai_image_url.startswith("http"): + openai_image_url = convert_url_to_base64(url=openai_image_url) + # Extract the media type and base64 data + media_type, base64_data = openai_image_url.split("data:")[1].split(";base64,") + + if format: + media_type = format + else: + media_type = media_type.replace("\\/", "/") + + return GenericImageParsingChunk( + type="base64", + media_type=media_type, + data=base64_data, + ) + except litellm.ImageFetchError: + raise + except Exception as e: + raise Exception( + f"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{{base64_image}}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']. Error: {e}""" + ) + + +def create_anthropic_image_param( + image_url_input: str | dict, + format: str | None = None, + is_bedrock_invoke: bool = False, +) -> AnthropicMessagesImageParam: + """ + Create an AnthropicMessagesImageParam from an image URL input. + + Supports both URL references (for HTTP/HTTPS URLs) and base64 encoding. + """ + # Extract URL and format from input + if isinstance(image_url_input, str): + image_url = image_url_input + else: + image_url = image_url_input.get("url", "") + if format is None: + format = image_url_input.get("format") + + # Check if the image URL is an HTTP/HTTPS URL + if image_url.startswith("http://") or image_url.startswith("https://"): + # For Bedrock invoke and Vertex AI Anthropic, always convert URLs to base64 + # as these providers don't support URL sources for images + if is_bedrock_invoke or image_url.startswith("http://"): + base64_url: Final = convert_url_to_base64(url=image_url) + image_chunk = convert_to_anthropic_image_obj(openai_image_url=base64_url, format=format) + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + else: + # HTTPS URL - pass directly for regular Anthropic + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSourceUrl( + type="url", + url=image_url, + ), + ) + else: + # Convert to base64 for data URIs or other formats + image_chunk = convert_to_anthropic_image_obj(openai_image_url=image_url, format=format) + return AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + + +# The following XML functions will be deprecated once JSON schema support is available on Bedrock and Vertex +# ------------------------------------------------------------------------------ +def convert_to_anthropic_tool_result_xml(message: dict) -> str: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + """ + + """ + Anthropic tool_results look like: + + [Successful results] + + + get_current_weather + + function result goes here + + + + + [Error results] + + + error message goes here + + + """ + name: Final = message.get("name") + content = message.get("content", "") + content = content.replace("<", "<").replace(">", ">").replace("&", "&") + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + anthropic_tool_result: Final = ( + "\n" + "\n" + f"{name}\n" + "\n" + f"{content}\n" + "\n" + "\n" + "" + ) + + return anthropic_tool_result + + +def convert_to_anthropic_tool_invoke_xml(tool_calls: list) -> str: + invokes = "" + for tool in tool_calls: + if get_attribute_or_key(tool, "type") != "function": + continue + + tool_function = get_attribute_or_key(tool, "function") + tool_name = get_attribute_or_key(tool_function, "name") + tool_arguments = get_attribute_or_key(tool_function, "arguments") + parsed_args = parse_tool_call_arguments( + tool_arguments, tool_name=tool_name, context="Anthropic XML tool invoke" + ) + if isinstance(parsed_args, dict): + parameters = "".join(f"<{param}>{val}\n" for param, val in parsed_args.items()) + else: + parameters = f"{parsed_args}\n" + invokes += f"\n{tool_name}\n\n{parameters}\n\n" + + anthropic_tool_invoke: Final = f"\n{invokes}" + + return anthropic_tool_invoke + + +def anthropic_messages_pt_xml(messages: list): + """ + format messages for anthropic + 1. Anthropic supports roles like "user" and "assistant", (here litellm translates system-> assistant) + 2. The first message always needs to be of role "user" + 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) + 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) + 5. System messages are a separate param to the Messages API (used for tool calling) + 6. Ensure we only accept role, content. (message.name is not supported) + """ + # add role=tool support to allow function call result/error submission + user_message_types: Final = {"user", "tool"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final = [] + msg_i = 0 + while msg_i < len(messages): + user_content = [] + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "image_url": + format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None + image_param = create_anthropic_image_param(m["image_url"], format=format) + # Convert to dict format for XML version + source = image_param["source"] + if isinstance(source, dict) and source.get("type") == "url": + # Type narrowing for URL source + url_source = cast(AnthropicContentParamSourceUrl, source) + user_content.append( + { + "type": "image", + "source": { + "type": "url", + "url": url_source["url"], + }, + } + ) + else: + # Type narrowing for base64 source + base64_source = cast(AnthropicContentParamSource, source) + user_content.append( + { + "type": "image", + "source": { + "type": "base64", + "media_type": base64_source["media_type"], + "data": base64_source["data"], + }, + } + ) + elif m.get("type", "") == "text": + user_content.append({"type": "text", "text": m["text"]}) + else: + # Tool message content will always be a string + user_content.append( + { + "type": "text", + "text": ( + convert_to_anthropic_tool_result_xml(messages[msg_i]) + if messages[msg_i]["role"] == "tool" + else messages[msg_i]["content"] + ), + } + ) + + msg_i += 1 + + if user_content: + new_messages.append({"role": "user", "content": user_content}) + + assistant_content = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_text = messages[msg_i].get("content") or "" # either string or none + if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion + assistant_text += convert_to_anthropic_tool_invoke_xml(messages[msg_i]["tool_calls"]) + + assistant_content.append({"type": "text", "text": assistant_text}) + msg_i += 1 + + if assistant_content: + new_messages.append({"role": "assistant", "content": assistant_content}) + + if not new_messages or new_messages[0]["role"] != "user": + if litellm.modify_params: + new_messages.insert(0, {"role": "user", "content": [{"type": "text", "text": "."}]}) + else: + raise Exception( + "Invalid first message. Should always start with 'role'='user' for Anthropic. System prompt is sent separately for Anthropic. set 'litellm.modify_params = True' or 'litellm_settings:modify_params = True' on proxy, to insert a placeholder user message - '.' as the first message, " + ) + + if new_messages[-1]["role"] == "assistant": + for content in new_messages[-1]["content"]: + if isinstance(content, dict) and content["type"] == "text": + content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message + + return new_messages + + +# ------------------------------------------------------------------------------ + + +def _azure_tool_call_invoke_helper( + function_call_params: ChatCompletionToolCallFunctionChunk, +) -> ChatCompletionToolCallFunctionChunk | None: + """ + Azure requires 'arguments' to be a string. + """ + if function_call_params.get("arguments") is None: + function_call_params["arguments"] = "" + return function_call_params + + +def _azure_image_url_helper(content: ChatCompletionImageObject): + if isinstance(content["image_url"], str): + content["image_url"] = {"url": content["image_url"]} + else: + content["image_url"] = cast( + ChatCompletionImageUrlObject, + {k: v for k, v in content["image_url"].items() if k != "format"}, + ) + + +def _azure_file_helper(content: ChatCompletionFileObject) -> None: + content["file"] = cast( + ChatCompletionFileObjectFile, + {k: v for k, v in content.get("file", {}).items() if k != "format"}, + ) + + +def convert_to_azure_openai_messages( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: + for m in messages: + if m["role"] == "assistant": + function_call = m.get("function_call", None) + if function_call is not None: + m["function_call"] = _azure_tool_call_invoke_helper(function_call) + + if m["role"] == "user" and isinstance(m.get("content"), list): + for content in m.get("content", []): + if isinstance(content, dict) and content.get("type") == "image_url": + _azure_image_url_helper(cast(ChatCompletionImageObject, content)) + elif isinstance(content, dict) and content.get("type") == "file": + _azure_file_helper(cast(ChatCompletionFileObject, content)) + return messages + + +# ------------------------------------------------------------------------------ + + +def infer_protocol_value( + value: object, +) -> Literal[ + "string_value", + "number_value", + "bool_value", + "struct_value", + "list_value", + "null_value", + "unknown", +]: + if value is None: + return "null_value" + if isinstance(value, int) or isinstance(value, float): + return "number_value" + if isinstance(value, str): + return "string_value" + if isinstance(value, bool): + return "bool_value" + if isinstance(value, dict): + return "struct_value" + if isinstance(value, list): + return "list_value" + + return "unknown" + + +def _gemini_tool_call_invoke_helper( + function_call_params: ChatCompletionToolCallFunctionChunk, + tool_call_id: str | None = None, +) -> VertexFunctionCall | None: + name: Final = function_call_params.get("name", "") or "" + arguments: Final = function_call_params.get("arguments", "") + if ( + isinstance(arguments, str) and len(arguments) == 0 + ): # pass empty dict, if arguments is empty string - prevents call from failing + arguments_dict = { + "type": "object", + } + else: + arguments_dict = json.loads(arguments) + function_call: Final = VertexFunctionCall( + name=name, + args=arguments_dict, + ) + if tool_call_id: + clean_id: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] + if clean_id: + function_call["id"] = clean_id + return function_call + + +def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: str | None) -> str: + """ + Embed thought signature into tool call ID for OpenAI client compatibility. + + Args: + tool_call_id: The tool call ID (e.g., "call_abc123...") + thought_signature: Base64-encoded signature from Gemini response + + Returns: + Tool call ID with embedded signature if present, otherwise original ID + Format: call___thought__ + + See: https://ai.google.dev/gemini-api/docs/thought-signatures + """ + if thought_signature: + return f"{tool_call_id}{THOUGHT_SIGNATURE_SEPARATOR}{thought_signature}" + return tool_call_id + + +def _get_thought_signature_from_tool(tool: dict) -> str | None: + """Extract thought signature from tool call's provider_specific_fields. + + If not provided try to extract thought signature from tool call id + + Checks both tool.provider_specific_fields and tool.function.provider_specific_fields. + Returns None when the tool call carries no signature; callers decide whether a + placeholder signature is needed. + """ + # First check tool's provider_specific_fields + provider_fields: Final = tool.get("provider_specific_fields") or {} + if isinstance(provider_fields, dict): + signature = provider_fields.get("thought_signature") + if signature: + return signature + + # Then check function's provider_specific_fields + function: Final = tool.get("function") + if function: + if isinstance(function, dict): + func_provider_fields: Final = function.get("provider_specific_fields") or {} + if isinstance(func_provider_fields, dict): + signature = func_provider_fields.get("thought_signature") + if signature: + return signature + elif hasattr(function, "provider_specific_fields") and function.provider_specific_fields: + if isinstance(function.provider_specific_fields, dict): + signature = function.provider_specific_fields.get("thought_signature") + if signature: + return signature + # Check if thought signature is embedded in tool call ID + tool_call_id: Final = tool.get("id") + if tool_call_id and THOUGHT_SIGNATURE_SEPARATOR in tool_call_id: + parts: Final = tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1) + if len(parts) == 2: + _, signature = parts + return signature + return None + + +def _get_dummy_thought_signature() -> str: + """Generate a dummy thought signature for models that require it. + + This is used when transferring conversation history from older models + (like gemini-2.5-flash) to gemini-3, which requires thought_signature + for strict validation. Google documents it as a last resort that "will + negatively impact model performance", so callers must only fall back to it + when no real signature is available. + + See: + https://ai.google.dev/gemini-api/docs/thought-signatures#faqs + https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures + """ + dummy_data: Final = b"skip_thought_signature_validator" + return base64.b64encode(dummy_data).decode("utf-8") + + +def convert_to_gemini_tool_call_invoke( + message: ChatCompletionAssistantMessage, + model: str | None = None, + forward_function_call_id: bool = False, +) -> list[VertexPartType]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + """ + Gemini tool call invokes: + { + "role": "model", + "parts": [ + { + "functionCall": { + "name": "get_current_weather", + "args": { + "unit": "fahrenheit", + "predicted_temperature": 45, + "location": "Boston, MA", + } + } + } + ] + } + """ + + """ + - json.load the arguments + """ + try: + _parts_list: Final[list[VertexPartType]] = [] + tool_calls: Final = message.get("tool_calls", None) + function_call: Final = message.get("function_call", None) + + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, + ) + + needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model) + + if tool_calls is not None: + for tool in tool_calls: + if "function" in tool: + gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper( + function_call_params=tool["function"], + tool_call_id=(tool.get("id") if forward_function_call_id else None), + ) + if gemini_function_call is not None: + part_dict: VertexPartType = {"function_call": gemini_function_call} + thought_signature = _get_thought_signature_from_tool(dict(tool)) + # Gemini signs only the first functionCall part of a parallel batch, so scope the + # placeholder fallback to that part instead of fabricating one per sibling call: + # https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example + is_first_function_call = len(_parts_list) == 0 + if not thought_signature and is_first_function_call and needs_dummy_signature: + thought_signature = _get_dummy_thought_signature() + if thought_signature: + part_dict["thoughtSignature"] = thought_signature + + _parts_list.append(part_dict) + else: # don't silently drop params. Make it clear to user what's happening. + raise Exception( + f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {tool}" + ) + elif function_call is not None: + gemini_function_call = _gemini_tool_call_invoke_helper(function_call_params=function_call) + if gemini_function_call is not None: + part_dict_function: Final[VertexPartType] = {"function_call": gemini_function_call} + + # Extract thought signature from function_call's provider_specific_fields + thought_signature = None + provider_fields: Final = ( + function_call.get("provider_specific_fields") if isinstance(function_call, dict) else {} + ) + if isinstance(provider_fields, dict): + thought_signature = provider_fields.get("thought_signature") + + # If no signature found and model is gemini-3, use dummy signature + if not thought_signature and needs_dummy_signature: + thought_signature = _get_dummy_thought_signature() + + if thought_signature: + part_dict_function["thoughtSignature"] = thought_signature + + _parts_list.append(part_dict_function) + else: # don't silently drop params. Make it clear to user what's happening. + raise Exception( + f"function_call missing. Received tool call with 'type': 'function'. No function call in argument - {message}" + ) + return _parts_list + except Exception as e: + raise Exception(f"Unable to convert openai tool calls={message} to gemini tool calls. Received error={e}") + + +def convert_to_gemini_tool_call_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + last_message_with_tool_calls: dict | None, + forward_function_call_id: bool = False, +) -> VertexPartType | list[VertexPartType]: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "content": "function result goes here", + }, + + # NOTE: Function messages have been deprecated + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + + Supports content with images for Computer Use: + { + "role": "tool", + "tool_call_id": "call_abc123", + "content": [ + {"type": "text", "text": "I found the requested image:"}, + {"type": "input_image", "image_url": "https://example.com/image.jpg" } + ] + } + """ + from litellm.types.llms.vertex_ai import BlobType + + content_str: str = "" + inline_data_list: Final[list[BlobType]] = [] + + if "content" in message: + if isinstance(message["content"], str): + content_str = message["content"] + # Detect data-URL images (e.g. from Anthropic tool_result with a single image block + # that was serialised as a plain string by translate_anthropic_messages_to_openai) + # and promote them to inline_data so Gemini receives actual image bytes. + if content_str[:5].lower() == "data:" and ";base64," in content_str: + try: + mime_rest: Final = content_str[5:].split(";base64,", 1) + if len(mime_rest) == 2 and mime_rest[0].startswith("image/"): + # Strip any extra parameters (e.g. ";charset=UTF-8") from the MIME segment + clean_mime: Final = mime_rest[0].split(";")[0].strip() + inline_data_list.append(BlobType(data=mime_rest[1], mime_type=clean_mime)) + content_str = "" + except Exception as e: + verbose_logger.warning("Failed to parse data URL in tool response: %s", e) + elif isinstance(message["content"], list): + content_list: Final = message["content"] + for content in content_list: + content_type = content.get("type", "") + if content_type == "text": + content_str += content.get("text", "") + elif content_type == "image": # pyright: ignore[reportUnnecessaryComparison] # loose runtime dict + # Anthropic-native image block: {"type": "image", "source": {"type": "base64", ...}} + source = content.get("source", {}) + if isinstance(source, dict) and source.get("type") == "base64": + try: + inline_data_list.append( + BlobType( + data=source.get("data", ""), + mime_type=source.get("media_type", "image/jpeg"), + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process Anthropic image block in tool response: %s", e) + elif content_type in ("input_image", "image_url"): + # Extract image for inline_data (for Computer Use screenshots and tool results) + image_url_data = content.get("image_url", "") + image_url = image_url_data.get("url", "") if isinstance(image_url_data, dict) else image_url_data + + if image_url: + # Convert image to base64 blob format for Gemini + try: + image_obj = convert_to_anthropic_image_obj(image_url, format=None) + inline_data_list.append( + BlobType( + data=image_obj["data"], + mime_type=image_obj["media_type"], + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process image in tool response: %s", e) + elif content_type in ("file", "input_file"): # pyright: ignore[reportUnnecessaryContains] # loose runtime dict + # Extract file for inline_data (for tool results with PDF, audio, video, etc.) + file_data = content.get("file_data", "") + if not file_data: + file_content = content.get("file", {}) + file_data = ( + file_content.get("file_data", "") + if isinstance(file_content, dict) + else file_content + if isinstance(file_content, str) + else "" + ) + + if file_data: + # Convert file to base64 blob format for Gemini + try: + file_obj = convert_to_anthropic_image_obj(file_data, format=None) + inline_data_list.append( + BlobType( + data=file_obj["data"], + mime_type=file_obj["media_type"], + ) + ) + except Exception as e: + verbose_logger.warning("Failed to process file in tool response: %s", e) + name: str | None = message.get("name", "") + + # Recover name from last message with tool calls + if last_message_with_tool_calls: + tools: Final = last_message_with_tool_calls.get("tool_calls", []) + msg_tool_call_id: Final = message.get("tool_call_id", None) + for tool in tools: + prev_tool_call_id = tool.get("id", None) + if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: + name = tool.get("function", {}).get("name", "") + + # Echo the OpenAI tool_call_id on functionResponse (strip thought-signature suffix). + gemini_call_id: str | None = None + if forward_function_call_id: + raw_tool_call_id: Final = message.get("tool_call_id") + if raw_tool_call_id and isinstance(raw_tool_call_id, str): + stripped_id: Final = raw_tool_call_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0] + if stripped_id: + gemini_call_id = stripped_id + + if not name: + raise Exception( + f"Missing corresponding tool call for tool response message. Received - message={message}, last_message_with_tool_calls={last_message_with_tool_calls}" + ) + + # Parse response data - support both JSON string and plain string + # For Computer Use, the response should contain structured data like {"url": "..."} + response_data: dict + try: + if content_str.strip().startswith("{") or content_str.strip().startswith("["): + # Try to parse as JSON (for Computer Use structured responses) + parsed: Final = json.loads(content_str) + if isinstance(parsed, dict): + response_data = parsed # Use the parsed JSON directly + else: + response_data = {"content": content_str} + else: + response_data = {"content": content_str} + except (json.JSONDecodeError, ValueError): + # Not valid JSON, wrap in content field + response_data = {"content": content_str} + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + _function_response: Final = VertexFunctionResponse( + name=name, + response=response_data, + ) + if gemini_call_id: + _function_response["id"] = gemini_call_id + + _part: Final[VertexPartType] = {"function_response": _function_response} + + # For multimodal function responses, Gemini expects media parts nested + # inside functionResponse.parts instead of sibling content parts. + if inline_data_list: + _function_response["parts"] = [{"inline_data": inline_data} for inline_data in inline_data_list] + return [_part] + + return _part + + +_TOOL_USE_ID_FALLBACK: Final = "tool_use_id" +_ANTHROPIC_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_-]") +_BEDROCK_TOOL_USE_ID_INVALID_CHARS: Final = re.compile(r"[^a-zA-Z0-9_.:-]") +_BEDROCK_TOOL_USE_ID_MAX_LEN: Final = 64 +_BEDROCK_TOOL_USE_ID_HASH_LEN: Final = 8 + + +def _replace_invalid_tool_use_id_chars(tool_use_id: str, invalid_chars: re.Pattern[str]) -> str: + return invalid_chars.sub("_", tool_use_id) or _TOOL_USE_ID_FALLBACK + + +def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str: + """Anthropic requires tool_use_id to match ^[a-zA-Z0-9_-]+$.""" + return _replace_invalid_tool_use_id_chars(tool_use_id, _ANTHROPIC_TOOL_USE_ID_INVALID_CHARS) + + +def _sanitize_bedrock_tool_use_id(tool_use_id: str) -> str: + """ + Bedrock Converse requires toolUseId to match [a-zA-Z0-9_.:-]+ and be at most 64 chars. + Ids that need rewriting get a short hash of the original appended so two ids that only + differ in a replaced char or past the cut still map to distinct values. + """ + sanitized: Final = _replace_invalid_tool_use_id_chars(tool_use_id, _BEDROCK_TOOL_USE_ID_INVALID_CHARS) + if sanitized == tool_use_id and len(sanitized) <= _BEDROCK_TOOL_USE_ID_MAX_LEN: + return sanitized + digest: Final = hashlib.sha256(tool_use_id.encode()).hexdigest()[:_BEDROCK_TOOL_USE_ID_HASH_LEN] + return f"{sanitized[: _BEDROCK_TOOL_USE_ID_MAX_LEN - _BEDROCK_TOOL_USE_ID_HASH_LEN - 1]}_{digest}" + + +_ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES: Final = {"application/pdf", "text/plain"} + + +def _is_anthropic_document_data_uri(url: str) -> bool: + # Anthropic's base64 document source accepts only application/pdf and + # text/plain (see select_anthropic_content_block_type_for_file). Routing + # other mimes here would produce a document block the API rejects, so we + # leave them on the image code path. + match: Final = re.match(r"data:([^;,]+)", url) + if not match: + return False + return match.group(1) in _ANTHROPIC_DOCUMENT_BASE64_MEDIA_TYPES + + +def convert_to_anthropic_tool_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + force_base64: bool = False, +) -> AnthropicMessagesToolResultParam: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + + """ + Anthropic tool_results look like: + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_01A09q90qw90lq917835lq9", + "content": "ConnectionError: the weather service API is not available (HTTP 500)", + # "is_error": true + } + ] + } + """ + anthropic_content: ( + str + | list[ + AnthropicMessagesToolResultContent + | AnthropicMessagesImageParam + | AnthropicMessagesDocumentParam + | ToolReference + ] + ) = "" + if isinstance(message["content"], str): + anthropic_content = message["content"] + elif isinstance(message["content"], list): + content_list: Final = message["content"] + anthropic_content_list: list[ + AnthropicMessagesToolResultContent + | AnthropicMessagesImageParam + | AnthropicMessagesDocumentParam + | ToolReference + ] = [] + for content in content_list: + if content["type"] == "text": + # Only include cache_control if explicitly set and not None + # to avoid sending "cache_control": null which breaks some API channels + text_content: AnthropicMessagesToolResultContent = { + "type": "text", + "text": content["text"], + } + cache_control_value = content.get("cache_control") + if cache_control_value is not None: + text_content["cache_control"] = cache_control_value + anthropic_content_list.append(text_content) + elif content["type"] == "image_url": + image_url_value = content["image_url"] + format = image_url_value.get("format") if isinstance(image_url_value, dict) else None + url_str = image_url_value.get("url") if isinstance(image_url_value, dict) else image_url_value + # Data URIs with non-image mime types (e.g. application/pdf) must + # translate to Anthropic document blocks, not image blocks — + # wrapping a PDF in `type: "image"` is rejected by the API. + if isinstance(url_str, str) and _is_anthropic_document_data_uri(url_str): + synth_file_message: ChatCompletionFileObject = { + "type": "file", + "file": {"file_data": url_str}, + } + _document_block = anthropic_process_openai_file_message(synth_file_message) + _document_block = add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, _document_block), + original_content_element=content, + ) + anthropic_content_list.append(cast(AnthropicMessagesDocumentParam, _document_block)) + else: + _anthropic_image_param = create_anthropic_image_param( + image_url_value, + format=format, + is_bedrock_invoke=force_base64, + ) + _anthropic_image_param = add_cache_control_to_content( + anthropic_content_element=_anthropic_image_param, + original_content_element=content, + ) + anthropic_content_list.append(cast(AnthropicMessagesImageParam, _anthropic_image_param)) + elif content["type"] == "tool_reference": + anthropic_content_list.append(ToolReference(type="tool_reference", tool_name=content["tool_name"])) + elif content["type"] == "file": + file_content = cast(ChatCompletionFileObject, content) + _file_block = anthropic_process_openai_file_message(file_content) + _file_block = add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, _file_block), + original_content_element=content, + ) + anthropic_content_list.append(_file_block) + + anthropic_content = anthropic_content_list + anthropic_tool_result: AnthropicMessagesToolResultParam | None = None + ## PROMPT CACHING CHECK ## + cache_control: Final = message.get("cache_control", None) + if message["role"] == "tool": + tool_message: Final[ChatCompletionToolMessage] = message + tool_call_id: str = tool_message["tool_call_id"] + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) + + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + anthropic_tool_result = AnthropicMessagesToolResultParam( + type="tool_result", + tool_use_id=sanitized_tool_use_id, + content=anthropic_content, + ) + + if message["role"] == "function": + function_message: Final[ChatCompletionFunctionMessage] = message + tool_call_id = function_message.get("tool_call_id") or str(uuid.uuid4()) + # Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$ + sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id) + anthropic_tool_result = AnthropicMessagesToolResultParam( + type="tool_result", + tool_use_id=sanitized_tool_use_id, + content=anthropic_content, + ) + + if anthropic_tool_result is None: + raise Exception(f"Unable to parse anthropic tool result for message: {message}") + if cache_control is not None: + anthropic_tool_result["cache_control"] = cache_control + return anthropic_tool_result + + +def convert_function_to_anthropic_tool_invoke( + function_call: dict | ChatCompletionToolCallFunctionChunk, +) -> list[AnthropicMessagesToolUseParam]: + try: + _name: Final = get_attribute_or_key(function_call, "name") or "" + _arguments: Final = get_attribute_or_key(function_call, "arguments") + + tool_input: Final = parse_tool_call_arguments( + _arguments, tool_name=_name, context="Anthropic function to tool invoke" + ) + + anthropic_tool_invoke: Final = [ + AnthropicMessagesToolUseParam( + type="tool_use", + id=str(uuid.uuid4()), + name=_name, + input=tool_input, + ) + ] + return anthropic_tool_invoke + except Exception as e: + raise e + + +def _find_server_tool_result( + tool_id: str, + web_search_results: Sequence[object] | None, + tool_results: Sequence[object] | None, +) -> dict[str, object] | None: + candidates: Final = (*(web_search_results or ()), *(tool_results or ())) + return next( + (result for result in candidates if isinstance(result, dict) and result.get("tool_use_id") == tool_id), + None, + ) + + +def convert_to_anthropic_tool_invoke( + tool_calls: list[ChatCompletionAssistantToolCall], + web_search_results: Sequence[object] | None = None, + tool_results: Sequence[object] | None = None, +) -> list[AnthropicMessagesToolUseParam | dict[str, Any]]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + + """ + Anthropic tool invokes: + { + "role": "assistant", + "content": [ + { + "type": "text", + "text": "To answer this question, I will: 1. Use the get_weather tool to get the current weather in San Francisco. 2. Use the get_time tool to get the current time in the America/Los_Angeles timezone, which covers San Francisco, CA." + }, + { + "type": "tool_use", + "id": "toolu_01A09q90qw90lq917835lq9", + "name": "get_weather", + "input": {"location": "San Francisco, CA"} + } + ] + } + + For server-side tools (web_search), we need to reconstruct: + - server_tool_use blocks (id starts with "srvtoolu_") + - web_search_tool_result blocks (from provider_specific_fields) + + Fixes: https://github.com/BerriAI/litellm/issues/17737 + """ + anthropic_tool_invoke: Final[list[AnthropicMessagesToolUseParam | dict[str, object]]] = [] + + for tool in tool_calls: + if get_attribute_or_key(tool, "type") != "function": + continue + + tool_id = cast(str, get_attribute_or_key(tool, "id")) + tool_name = cast( + str, + get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), + ) + tool_input = parse_tool_call_arguments( + get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments"), + tool_name=tool_name, + context="Anthropic tool invoke", + ) + + server_tool_result = ( + _find_server_tool_result(tool_id, web_search_results, tool_results) + if tool_id.startswith("srvtoolu_") + else None + ) + if server_tool_result is not None: + anthropic_tool_invoke.append( + { + "type": "server_tool_use", + "id": tool_id, + "name": tool_name, + "input": tool_input, + } + ) + anthropic_tool_invoke.append(server_tool_result) + else: + sanitized_tool_id = _sanitize_anthropic_tool_use_id(tool_id) + _anthropic_tool_use_param = AnthropicMessagesToolUseParam( + type="tool_use", + id=sanitized_tool_id, + name=tool_name, + input=tool_input, + ) + + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_tool_use_param, + original_content_element=dict(tool), + ) + + if "cache_control" in _content_element: + _anthropic_tool_use_param["cache_control"] = _content_element["cache_control"] + + anthropic_tool_invoke.append(_anthropic_tool_use_param) + + return anthropic_tool_invoke + + +def add_cache_control_to_content( + anthropic_content_element: dict + | AnthropicMessagesImageParam + | AnthropicMessagesTextParam + | AnthropicMessagesDocumentParam + | AnthropicMessagesToolUseParam + | ChatCompletionThinkingBlock, + original_content_element: dict | AllMessageValues, +): + cache_control_param: Final = original_content_element.get("cache_control") + if cache_control_param is not None and isinstance(cache_control_param, dict): + transformed_param: Final = ChatCompletionCachedContent(**cache_control_param) + + anthropic_content_element["cache_control"] = transformed_param + + return anthropic_content_element + + +def _anthropic_content_element_factory( + image_chunk: GenericImageParsingChunk, +) -> AnthropicMessagesImageParam | AnthropicMessagesDocumentParam: + if image_chunk["media_type"] == "application/pdf": + _anthropic_content_element: AnthropicMessagesDocumentParam | AnthropicMessagesImageParam = ( + AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + ) + else: + _anthropic_content_element = AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + + return _anthropic_content_element + + +def select_anthropic_content_block_type_for_file( + format: str, +) -> Literal["document", "image", "container_upload"]: + if format == "application/pdf" or format == "text/plain": + return "document" + elif format in ["image/jpeg", "image/png", "image/gif", "image/webp"]: + return "image" + else: + return "container_upload" + + +def anthropic_infer_file_id_content_type( + file_id: str, +) -> Literal["document_url", "container_upload"]: + """ + Use when 'format' not provided. + + - URL's - assume are document_url + - Else - assume is container_upload + """ + if file_id.startswith("http") or file_id.startswith("https"): + return "document_url" + else: + return "container_upload" + + +def anthropic_process_openai_file_message( + message: ChatCompletionFileObject, +) -> AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam: + file_message: Final = cast(ChatCompletionFileObject, message) + file_sub: Final = file_message.get("file") + if file_sub is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="anthropic", + ) + file_data: Final = file_sub.get("file_data") + file_id: Final = file_sub.get("file_id") + format: Final = file_sub.get("format") + if file_data: + image_chunk: Final = convert_to_anthropic_image_obj( + openai_image_url=file_data, + format=format, + ) + anthropic_document_param: Final = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSource( + type="base64", + media_type=image_chunk["media_type"], + data=image_chunk["data"], + ), + ) + return anthropic_document_param + elif file_id: + content_block_type: Final = ( + select_anthropic_content_block_type_for_file(format) + if format + else anthropic_infer_file_id_content_type(file_id) + ) + return_block_param: ( + AnthropicMessagesDocumentParam | AnthropicMessagesImageParam | AnthropicMessagesContainerUploadParam | None + ) = None + if content_block_type == "document": + return_block_param = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSourceFileId( + type="file", + file_id=file_id, + ), + ) + elif content_block_type == "document_url": + return_block_param = AnthropicMessagesDocumentParam( + type="document", + source=AnthropicContentParamSourceUrl( + type="url", + url=file_id, + ), + ) + elif content_block_type == "image": + return_block_param = AnthropicMessagesImageParam( + type="image", + source=AnthropicContentParamSourceFileId( + type="file", + file_id=file_id, + ), + ) + elif content_block_type == "container_upload": + return_block_param = AnthropicMessagesContainerUploadParam(type="container_upload", file_id=file_id) + + if return_block_param is None: + raise Exception(f"Unable to parse anthropic file message: {message}") + return return_block_param + raise Exception(f"Either file_data or file_id must be present in the file message: {message}") + + +_EMPTY_TEXT_PLACEHOLDER: Final = "[System: Empty message content sanitised to satisfy protocol]" + + +def _sanitize_empty_text_content( + message: AllMessageValues, +) -> AllMessageValues: + """ + Case C: Sanitize empty text content + - Replace empty or whitespace-only text content with a placeholder message. + - Handles both string content and list-of-blocks content (rewriting only + the empty text blocks in place; non-text blocks like images are left + untouched). + + Returns: + The message with sanitized content if needed, otherwise the original message + """ + if message.get("role") not in ["user", "assistant"]: + return message + + content: Final = message.get("content") + + if isinstance(content, str): + if not content or not content.strip(): + message = cast(AllMessageValues, dict(message)) # Make a copy + message["content"] = _EMPTY_TEXT_PLACEHOLDER + verbose_logger.debug( + "_sanitize_empty_text_content: Replaced empty text content in %s message", message.get("role") + ) + return message + + if isinstance(content, list): + # Walk the blocks and rewrite any empty text blocks. We rewrite (rather + # than drop) so callers don't end up with an entirely empty content + # list, which Anthropic also rejects. + new_blocks: Final[list[Any]] = [] + rewrote_any = False + for block in content: + if isinstance(block, dict) and block.get("type") == "text": + text = block.get("text") + if not isinstance(text, str) or not text or not text.strip(): + new_block = dict(block) + new_block["text"] = _EMPTY_TEXT_PLACEHOLDER + new_blocks.append(new_block) + rewrote_any = True + continue + new_blocks.append(block) + + if rewrote_any: + message = cast(AllMessageValues, dict(message)) # Make a copy + message["content"] = new_blocks + verbose_logger.debug( + "_sanitize_empty_text_content: Replaced empty text block(s) in %s message", message.get("role") + ) + + return message + + +def _add_missing_tool_results( + current_message: AllMessageValues, + messages: list[AllMessageValues], + current_index: int, +) -> tuple[list[AllMessageValues], int]: + """ + Case A: Missing tool_result for tool_use (orphaned tool calls) + - If an assistant message has tool_calls but no corresponding tool result follows, + add a dummy tool result message indicating the user did not provide the result. + + Returns: + A tuple of: + - List containing the assistant message, followed by existing tool results, + followed by any dummy tool results needed + - Number of original messages consumed (to adjust iteration index) + """ + result_messages: Final[list[AllMessageValues]] = [] + tool_calls: Final = current_message.get("tool_calls") + + if not tool_calls or len(cast(list, tool_calls)) == 0: + return ([current_message], 0) + + # Collect all tool_call_ids from this assistant message + expected_tool_call_ids: Final = set() + for tool_call in cast(list, tool_calls): + tool_call_id = None + if isinstance(tool_call, dict): + tool_call_id = tool_call.get("id") + else: + tool_call_id = getattr(tool_call, "id", None) + if tool_call_id: + expected_tool_call_ids.add(tool_call_id) + + # Collect actual tool result messages that follow this assistant message + found_tool_call_ids: Final = set() + actual_tool_results: Final[list[AllMessageValues]] = [] + j = current_index + 1 + + while j < len(messages): + next_msg = messages[j] + next_role = next_msg.get("role") + + if next_role == "assistant": + break + + if next_role in ["tool", "function"]: + tool_call_id = next_msg.get("tool_call_id") + if tool_call_id and tool_call_id in expected_tool_call_ids: + found_tool_call_ids.add(tool_call_id) + actual_tool_results.append(next_msg) + + j += 1 + + # Find missing tool results + missing_tool_call_ids: Final = expected_tool_call_ids - found_tool_call_ids + + if missing_tool_call_ids: + verbose_logger.debug( + "_add_missing_tool_results: Found %s orphaned tool calls. Adding dummy tool results.", + len(missing_tool_call_ids), + ) + + result_messages.append(current_message) + + # Add existing tool results FIRST + result_messages.extend(actual_tool_results) + + # Then add dummy tool results for missing ones + for tool_call_id in missing_tool_call_ids: + tool_name = "unknown_tool" + for tool_call in cast(list, tool_calls): + tc_id = None + if isinstance(tool_call, dict): + tc_id = tool_call.get("id") + else: + tc_id = getattr(tool_call, "id", None) + + if tc_id == tool_call_id: + if isinstance(tool_call, dict): + function = tool_call.get("function", {}) + if isinstance(function, dict): + tool_name = function.get("name", "unknown_tool") + else: + tool_name = getattr(function, "name", "unknown_tool") + else: + function = getattr(tool_call, "function", None) + if function: + tool_name = getattr(function, "name", "unknown_tool") + break + + dummy_tool_result: ChatCompletionToolMessage = { + "role": "tool", + "tool_call_id": tool_call_id, + "content": f"[System: Tool execution skipped/interrupted by user. No result provided for tool '{tool_name}'.]", + } + result_messages.append(dummy_tool_result) + + # Return the messages and the number of original messages to skip + return (result_messages, len(actual_tool_results)) + + return ([current_message], 0) + + +def _is_orphaned_tool_result( + current_message: AllMessageValues, + sanitized_messages: list[AllMessageValues], +) -> bool: + """ + Case B: Orphaned tool_result (unexpected result) + - Check if a tool message references a tool_call_id that doesn't exist in the previous + assistant message. + + Returns: + True if this is an orphaned tool result that should be removed, False otherwise + """ + if current_message.get("role") not in ["tool", "function"]: + return False + + tool_call_id: Final = current_message.get("tool_call_id") + + if not tool_call_id: + return False + + # Look back to find the most recent assistant message with tool_calls + found_matching_tool_call = False + + for j in range(len(sanitized_messages) - 1, -1, -1): + prev_msg = sanitized_messages[j] + if prev_msg.get("role") == "assistant": + tool_calls = prev_msg.get("tool_calls") + if tool_calls: + for tool_call in cast(list, tool_calls): + tc_id = None + if isinstance(tool_call, dict): + tc_id = tool_call.get("id") + else: + tc_id = getattr(tool_call, "id", None) + + if tc_id == tool_call_id: + found_matching_tool_call = True + break + + break + + if not found_matching_tool_call: + verbose_logger.debug("_is_orphaned_tool_result: Found orphaned tool result with redacted tool_call_id") + return True + + return False + + +def _declared_tool_call_ids(message: Mapping[str, object]) -> frozenset[str]: + tool_calls: Final = message.get("tool_calls") + if not isinstance(tool_calls, list): + return frozenset() + return frozenset( + str(tool_call["id"]) for tool_call in tool_calls if isinstance(tool_call, Mapping) and tool_call.get("id") + ) + + +def group_tool_exchanges(messages: Sequence[Mapping[str, object]]) -> tuple[tuple[int, ...], ...]: + """Group message indices into tool exchanges: an assistant row that made + tool calls, together with the tool rows answering the ids it declared. + + Membership is by ``tool_call_id`` ownership rather than adjacency, so a tool + row belonging to some other call opens its own group instead of being swept + into the exchange it happens to sit next to. Every other row is its own + group. Groups stay contiguous and in order, so a caller can convert or + protect them without reordering the conversation. + + Callers need this because an assistant row and the tool rows answering it + are only well-formed together: ``sanitize_messages_for_tool_calling`` reads + an assistant row whose results are missing as an orphaned tool call, and + a tool row whose call is missing as an orphaned result. + """ + return tuple(_iter_tool_exchange_groups(messages)) + + +def _iter_tool_exchange_groups(messages: Sequence[Mapping[str, object]]) -> Iterator[tuple[int, ...]]: + index = 0 + while index < len(messages): + declared = _declared_tool_call_ids(messages[index]) + end = index + 1 + while ( + declared + and end < len(messages) + and messages[end].get("role") in ("tool", "function") + and str(messages[end].get("tool_call_id")) in declared + ): + end += 1 + yield tuple(range(index, end)) + index = end + + +def sanitize_messages_for_tool_calling( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: + """ + Sanitize messages for tool calling to handle common issues when modify_params=True: + + Case A: Missing tool_result for tool_use (orphaned tool calls) + - If an assistant message has tool_calls but no corresponding tool result follows, + add a dummy tool result message indicating the user did not provide the result. + + Case B: Orphaned tool_result (unexpected result) + - If a tool message references a tool_call_id that doesn't exist in the previous + assistant message, remove that tool message. + + Case C: Empty text content + - Replace empty or whitespace-only text content with a placeholder message. + + Case D: Duplicate tool_result for same tool_use (duplicate results) + - If multiple tool messages reference the same tool_call_id, keep only the last + occurrence. Anthropic requires exactly one tool_result per tool_use and rejects + with: "each tool_use must have a single result". + + This function operates on OpenAI format messages before they are converted to + provider-specific formats. + """ + if not litellm.modify_params: + return messages + + sanitized_messages: list[AllMessageValues] = [] + i = 0 + + while i < len(messages): + current_message = messages[i] + + # Case C: Sanitize empty text content + current_message = _sanitize_empty_text_content(current_message) + + # Case A: Check if assistant message has tool_calls without following tool results + if current_message.get("role") == "assistant": + result_messages, messages_consumed = _add_missing_tool_results(current_message, messages, i) + + # If dummy tool results were added, extend sanitized_messages and skip consumed messages + if len(result_messages) > 1: + sanitized_messages.extend(result_messages) + # Skip the assistant message and any actual tool results that were included + i += 1 + messages_consumed + continue + + # Case B: Check for orphaned tool results + if _is_orphaned_tool_result(current_message, sanitized_messages): + i += 1 + continue # Skip this orphaned tool result + + # Add the message to sanitized list + sanitized_messages.append(current_message) + i += 1 + + # Case D: Deduplicate tool results with the same tool_call_id. + # Anthropic requires exactly one tool_result per tool_use. Session history + # (e.g. from conversation resume) can contain duplicate tool_result messages + # for the same tool_call_id. Keep only the last occurrence *within each + # contiguous block of tool results following an assistant message*. This + # avoids dropping results from earlier turns if a tool_call_id is reused. + # + # NOTE: This intentionally keeps the *last* occurrence (most complete for + # session-resume duplicates), unlike _deduplicate_bedrock_content_blocks + # which keeps the *first*. The Bedrock case handles provider-side content + # block duplication where the first is authoritative; here the duplicate + # arises from history replay where the last entry is the final state. + duplicates_to_remove: Final[set[int]] = set() + seen_in_block: dict[str, int] = {} # tool_call_id -> index (reset per block) + for idx, msg in enumerate(sanitized_messages): + role = msg.get("role") + tcid = msg.get("tool_call_id") if role in ["tool", "function"] else None + if tcid and isinstance(tcid, str): + if tcid in seen_in_block: + # Mark the earlier occurrence for removal (keep latest) + duplicates_to_remove.add(seen_in_block[tcid]) + verbose_logger.warning( + "sanitize_messages_for_tool_calling: dropping duplicate " + "tool_result with tool_call_id=%s. This may indicate " + "duplicate tool messages in conversation history.", + tcid, + ) + seen_in_block[tcid] = idx + elif role not in ("tool", "function"): + # Non-tool message (user, assistant, system) marks a + # conversational-turn boundary — reset tracking. + # Tool/function messages with no tool_call_id are malformed; + # they should NOT reset the block because they don't represent + # a turn boundary and would mask real within-block duplicates. + seen_in_block = {} + + if duplicates_to_remove: + sanitized_messages = [msg for idx, msg in enumerate(sanitized_messages) if idx not in duplicates_to_remove] + + return sanitized_messages + + +def _drop_unsignable_thinking_blocks( + thinking_blocks: list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], +) -> list[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock]: + return [block for block in thinking_blocks if not is_unsignable_thinking_block(block)] + + +_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues] + + +def anthropic_messages_pt( + messages: list[AllMessageValues], + model: str, + llm_provider: str, +) -> _AnthropicMessageList: + """ + format messages for anthropic + 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately). + Models flagged ``supports_mid_conversation_system`` also accept "system" inside + messages after a user turn; the caller decides placement, this keeps such messages. + 2. The first message always needs to be of role "user" + 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) + 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) + 5. System messages are a separate param to the Messages API + 6. Ensure we only accept role, content. (message.name is not supported) + """ + # Sanitize messages for tool calling issues when modify_params=True + messages = sanitize_messages_for_tool_calling(messages) + + # Anthropic rejects empty text content blocks with: + # "messages: text content blocks must be non-empty" + # OpenAI/other providers silently tolerate `{"role": "user", "content": ""}`, + # so callers (and upstream agent frameworks like pydantic-ai) routinely + # send empty user/assistant turns. We always rewrite these to a placeholder + # for Anthropic-shaped requests, independent of `litellm.modify_params`, + # because there is no way to "pass through" an empty text block — the + # request will always 400 otherwise. The richer tool-call sanitization + # (Cases A/B/D in `sanitize_messages_for_tool_calling`) remains gated on + # `modify_params` because it actually mutates conversation structure. + messages = [_sanitize_empty_text_content(m) for m in messages] + + # add role=tool support to allow function call result/error submission + user_message_types: Final = {"user", "tool", "function"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract + + if len(messages) == 0: + if not litellm.modify_params: + raise litellm.BadRequestError( + message=f"Anthropic requires at least one non-system message. Either provide one, or set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add the dummy user message - {DEFAULT_USER_CONTINUE_MESSAGE_TYPED}.", + model=model, + llm_provider=llm_provider, + ) + else: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE_TYPED) + + # Bedrock invoke models have format: invoke/... + # Vertex AI Anthropic also doesn't support URL sources for images + is_bedrock_invoke = model.lower().startswith("invoke/") + is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False + force_base64 = is_bedrock_invoke or is_vertex_ai + + msg_i = 0 + while msg_i < len(messages): + user_content: list[AnthropicMessagesUserMessageValues] = [] + init_msg_i = msg_i + if isinstance(messages[msg_i], BaseModel): + messages[msg_i] = dict(messages[msg_i]) + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + user_message_types_block: ( + ChatCompletionToolMessage | ChatCompletionUserMessage | ChatCompletionFunctionMessage + ) = messages[msg_i] + if user_message_types_block["role"] == "user": + if isinstance(user_message_types_block["content"], list): + user_content_start: int = len(user_content) # pyright: ignore[reportGeneralTypeIssues] # reset per user message + for m in user_message_types_block["content"]: + if m.get("type", "") == "image_url": + m = cast(ChatCompletionImageObject, m) + format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None + # Convert ChatCompletionImageUrlObject to dict if needed + image_url_value = m["image_url"] + if isinstance(image_url_value, str): + image_url_input: str | dict[str, object] = image_url_value + else: + # ChatCompletionImageUrlObject or dict case - convert to dict + image_url_input = { + "url": image_url_value["url"], + "format": image_url_value.get("format"), + } + # Bedrock invoke models have format: invoke/... + # Vertex AI Anthropic also doesn't support URL sources for images + is_bedrock_invoke = model.lower().startswith("invoke/") + is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False + force_base64 = is_bedrock_invoke or is_vertex_ai + _anthropic_content_element = create_anthropic_image_param( + image_url_input, + format=format, + is_bedrock_invoke=force_base64, + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_content_element, + original_content_element=dict(m), + ) + + if "cache_control" in _content_element: + _anthropic_content_element["cache_control"] = _content_element["cache_control"] + user_content.append(_anthropic_content_element) + elif m.get("type", "") == "text": + m = cast(ChatCompletionTextObject, m) + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=m["text"], + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(m), + ) + _content_element = cast(AnthropicMessagesTextParam, _content_element) + + user_content.append(_content_element) + elif m.get("type", "") == "document": + _document_content_element = cast( + AnthropicMessagesDocumentParam, + add_cache_control_to_content( + anthropic_content_element=cast(AnthropicMessagesDocumentParam, m), + original_content_element=dict(m), + ), + ) + user_content.append(_document_content_element) + elif m.get("type", "") == "file": + _file_content_element = anthropic_process_openai_file_message( + cast(ChatCompletionFileObject, m) + ) + _file_content_element = add_cache_control_to_content( + anthropic_content_element=cast( + AnthropicMessagesDocumentParam, + _file_content_element, + ), + original_content_element=dict(m), + ) + user_content.append( + cast( + AnthropicMessagesDocumentParam, + _file_content_element, + ) + ) + if ( + len(user_content) > user_content_start + and user_message_types_block.get("cache_control") is not None + and not any( + "cache_control" in content and content["cache_control"] is not None + for content in user_content[user_content_start:] + ) + ): + cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) # pyright: ignore[reportGeneralTypeIssues] # fresh copy per user message + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], cached_user_content + ), + original_content_element=user_message_types_block, + ) + user_content[-1] = cached_user_content + elif isinstance(user_message_types_block["content"], str): + _anthropic_content_text_element: AnthropicMessagesTextParam = { + "type": "text", + "text": user_message_types_block["content"], + } + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_content_text_element, + original_content_element=dict(user_message_types_block), + ) + + if "cache_control" in _content_element: + _anthropic_content_text_element["cache_control"] = _content_element["cache_control"] + + user_content.append(_anthropic_content_text_element) + + elif user_message_types_block["role"] == "tool" or user_message_types_block["role"] == "function": + # OpenAI's tool message content will always be a string + user_content.append( + convert_to_anthropic_tool_result(user_message_types_block, force_base64=force_base64) + ) + + msg_i += 1 + + if user_content: + new_messages.append({"role": "user", "content": user_content}) + + # Track unique tool IDs in this merge block to avoid duplication + unique_tool_ids: set[str] = set() + + assistant_content: list[AnthropicMessagesAssistantMessageValues] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_content_block: ChatCompletionAssistantMessage = messages[msg_i] + + # Extract compaction_blocks from provider_specific_fields and add them first + _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") + if isinstance(_provider_specific_fields_raw, dict): + _compaction_blocks = _provider_specific_fields_raw.get("compaction_blocks") + if _compaction_blocks and isinstance(_compaction_blocks, list): + # Add compaction blocks at the beginning of assistant content : https://platform.claude.com/docs/en/build-with-claude/compaction + assistant_content.extend(_compaction_blocks) + + _raw_thinking_blocks = assistant_content_block.get("thinking_blocks", None) + thinking_blocks = ( + _drop_unsignable_thinking_blocks(_raw_thinking_blocks) if _raw_thinking_blocks is not None else None + ) + + # Check if tool_calls contain server tool calls (web search, etc.) + # If so, we need to interleave thinking blocks with tool call groups + # to preserve the original content block ordering. + # Fixes: https://github.com/BerriAI/litellm/issues/23047 + assistant_tool_calls = assistant_content_block.get("tool_calls") + _has_server_tool_calls = False + if assistant_tool_calls is not None: + for _tc in assistant_tool_calls: + _tc_id = _tc.get("id") if isinstance(_tc, dict) else getattr(_tc, "id", None) + if _tc_id and isinstance(_tc_id, str) and _tc_id.startswith("srvtoolu_"): + _has_server_tool_calls = True + break + + _content_is_list: bool = "content" in assistant_content_block and isinstance( + assistant_content_block["content"], list + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + _content_list = assistant_content_block.get("content") if _content_is_list else None # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + _list_has_thinking: bool = ( + _content_is_list + and _content_list is not None + and any( + isinstance(item, dict) + and cast( # cast-ok: list items are inspected only for their type discriminator + dict[str, object], item + ).get("type") + in ("thinking", "redacted_thinking") + for item in _content_list + ) + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + assistant_content_start: int = len(assistant_content) + ( + len(thinking_blocks) + if _content_is_list and thinking_blocks is not None and not _list_has_thinking + else 0 + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + + if ( + thinking_blocks is not None + and _has_server_tool_calls + and isinstance(assistant_content_block.get("content", None), (str, type(None))) + ): + # INTERLEAVED MODE: When we have both thinking blocks and server + # tool calls (e.g. web search), Anthropic's original response + # interleaves them: [thinking_1, server_tool_use_1, result_1, + # thinking_2, text, server_tool_use_2, result_2, ...]. + # We must preserve this interleaved order because Anthropic + # verifies thinking block signatures based on position. + + # Build the tool call groups (server_tool_use + its result) + _provider_specific_fields_raw_tc = assistant_content_block.get("provider_specific_fields") + _provider_specific_fields_tc: dict[str, Any] = {} + if isinstance(_provider_specific_fields_raw_tc, dict): + _provider_specific_fields_tc = cast(dict[str, Any], _provider_specific_fields_raw_tc) + _web_search_results_tc = _provider_specific_fields_tc.get("web_search_results") + _tool_results_tc = _provider_specific_fields_tc.get("tool_results") + tool_invoke_results = convert_to_anthropic_tool_invoke( + assistant_tool_calls, + web_search_results=_web_search_results_tc, + tool_results=_tool_results_tc, + ) + + # Group tool invoke results into (server_tool_use, result) pairs + # and separate regular tool_use blocks + server_tool_groups: list[list[Any]] = [] + regular_tool_uses: list[Any] = [] + _current_group: list[Any] = [] + for item in tool_invoke_results: + item_type = item.get("type", "") if isinstance(item, dict) else getattr(item, "type", "") + if item_type == "server_tool_use": + if _current_group: + server_tool_groups.append(_current_group) + _current_group = [item] + elif item_type.endswith("_tool_result"): + _current_group.append(item) + elif item_type == "tool_use": + regular_tool_uses.append(item) + else: + _current_group.append(item) + if _current_group: + server_tool_groups.append(_current_group) + + # Build the text block if content is a non-empty string + text_element = None + _acb_content = assistant_content_block.get("content") + if isinstance(_acb_content, str) and _acb_content: + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=_acb_content, + ) + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(assistant_content_block), + ) + if "cache_control" in _content_element: + _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] + text_element = _anthropic_text_content_element + + # Interleave: each thinking block precedes its server tool group. + # Pattern: thinking[0], group[0], thinking[1], group[1], ... + # Any remaining thinking blocks (after all groups) go before text. + # Any remaining groups (after all thinking blocks) go after. + tb_idx = 0 + grp_idx = 0 + num_tb = len(thinking_blocks) if thinking_blocks else 0 + num_grp = len(server_tool_groups) + + while tb_idx < num_tb or grp_idx < num_grp: + if tb_idx < num_tb and grp_idx < num_grp: + # Emit thinking block then its tool group + assistant_content.append(thinking_blocks[tb_idx]) + tb_idx += 1 + for block in server_tool_groups[grp_idx]: + item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) + grp_idx += 1 + elif tb_idx < num_tb: + # More thinking blocks than tool groups - emit before text + assistant_content.append(thinking_blocks[tb_idx]) + tb_idx += 1 + else: + # More tool groups than thinking blocks - emit remaining + for block in server_tool_groups[grp_idx]: + item_id = block.get("id") if isinstance(block, dict) else getattr(block, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, block)) + grp_idx += 1 + + # Add text block (if any) + if text_element is not None: + assistant_content.append(text_element) + + # Add regular (non-server) tool calls at the end + for item in regular_tool_uses: + item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) + if item_id and item_id in unique_tool_ids: + continue + if item_id: + unique_tool_ids.add(item_id) + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) + + # Mark tool_calls as already processed so they are not added again + assistant_tool_calls = None + + else: + # SEQUENTIAL MODE: No server tool calls, or no thinking blocks, + # or content is a list. Use the original sequential approach. + + # When content is a list, check if it already contains thinking + # blocks inline. If so, skip prepending thinking_blocks to avoid + # duplication and preserve the original interleaved order. + # Fixes the gap where list-content messages bypass INTERLEAVED + # MODE and still get thinking blocks prepended out of order. + if ( + thinking_blocks is not None and not _list_has_thinking + ): # IMPORTANT: ADD THIS FIRST, ELSE ANTHROPIC WILL RAISE AN ERROR + assistant_content.extend(thinking_blocks) + if _content_is_list and _content_list is not None: + for m in _content_list: + if not isinstance(m, dict): + continue + # handle thinking blocks + thinking_block = cast(str, m.get("thinking", "")) + text_block = cast(str, m.get("text", "")) + if ( + m.get("type", "") == "thinking" + and len(thinking_block) > 0 + and not is_unsignable_thinking_block(m) + ): # don't pass empty text blocks. anthropic api raises errors. + anthropic_message: ChatCompletionThinkingBlock | AnthropicMessagesTextParam = cast( + ChatCompletionThinkingBlock, m + ) + assistant_content.append(anthropic_message) + # handle text + elif ( + m.get("type", "") == "text" and len(text_block) > 0 + ): # don't pass empty text blocks. anthropic api raises errors. + anthropic_message = AnthropicMessagesTextParam(type="text", text=text_block) + _cached_message = add_cache_control_to_content( + anthropic_content_element=anthropic_message, + original_content_element=dict(m), + ) + + assistant_content.append(cast(AnthropicMessagesTextParam, _cached_message)) + # handle server_tool_use blocks (tool search, web search, etc.) + # Pass through as-is since these are Anthropic-native content types + elif m.get("type", "") == "server_tool_use" or m.get("type", "").endswith("_tool_result"): + assistant_content.append(m) + elif ( + "content" in assistant_content_block + and isinstance(assistant_content_block["content"], str) + and assistant_content_block["content"] # don't pass empty text blocks. anthropic api raises errors. + ): + _anthropic_text_content_element = AnthropicMessagesTextParam( + type="text", + text=assistant_content_block["content"], + ) + + _content_element = add_cache_control_to_content( + anthropic_content_element=_anthropic_text_content_element, + original_content_element=dict(assistant_content_block), + ) + + if "cache_control" in _content_element: + _anthropic_text_content_element["cache_control"] = _content_element["cache_control"] + + assistant_content.append(_anthropic_text_content_element) + + if assistant_tool_calls is not None: # support assistant tool invoke conversion + # Get web_search_results and tool_results from provider_specific_fields + # for server_tool_use reconstruction. + # Fixes: https://github.com/BerriAI/litellm/issues/17737 + _provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields") + _provider_specific_fields: dict[str, Any] = {} + if isinstance(_provider_specific_fields_raw, dict): + _provider_specific_fields = cast(dict[str, Any], _provider_specific_fields_raw) + _web_search_results = _provider_specific_fields.get("web_search_results") + _tool_results = _provider_specific_fields.get("tool_results") + tool_invoke_results = convert_to_anthropic_tool_invoke( + assistant_tool_calls, + web_search_results=_web_search_results, + tool_results=_tool_results, + ) + + # Prevent "tool_use ids must be unique" errors by filtering duplicates + # This can happen when merging history that already contains the tool calls + for item in tool_invoke_results: + # tool_use items are typically dicts, but handle objects just in case + item_id = item.get("id") if isinstance(item, dict) else getattr(item, "id", None) + + if item_id: + if item_id in unique_tool_ids: + continue + unique_tool_ids.add(item_id) + + assistant_content.append(cast(AnthropicMessagesAssistantMessageValues, item)) + + assistant_function_call = assistant_content_block.get("function_call") + + if assistant_function_call is not None: + assistant_content.extend(convert_function_to_anthropic_tool_invoke(assistant_function_call)) + + if ( + _content_is_list + and len(assistant_content) > assistant_content_start + and assistant_content_block.get("cache_control") is not None + and not any( + "cache_control" in content and content["cache_control"] is not None + for content in assistant_content[assistant_content_start:] + ) + ): + cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) # pyright: ignore[reportGeneralTypeIssues] # fresh copy per assistant message + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], cached_assistant_content + ), + original_content_element=assistant_content_block, + ) + assistant_content[-1] = cached_assistant_content + + msg_i += 1 + + if assistant_content: + new_messages.append({"role": "assistant", "content": assistant_content}) + + ## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + new_messages.extend(anthropic_system_messages(messages[msg_i])) + msg_i += 1 + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + if len(new_messages) > 0 and new_messages[-1]["role"] == "assistant": + if isinstance(new_messages[-1]["content"], str): + new_messages[-1]["content"] = new_messages[-1]["content"].rstrip() + elif isinstance(new_messages[-1]["content"], list): + for content in new_messages[-1]["content"]: + if isinstance(content, dict) and content["type"] == "text": + content["text"] = content["text"].rstrip() # no trailing whitespace for final assistant message + + return new_messages + + +def extract_between_tags(tag: str, string: str, strip: bool = False) -> list[str]: + ext_list = re.findall(f"<{tag}>(.+?)", string, re.DOTALL) + if strip: + ext_list = [e.strip() for e in ext_list] + return ext_list + + +def contains_tag(tag: str, string: str) -> bool: + return bool(re.search(f"<{tag}>(.+?)", string, re.DOTALL)) + + +def parse_xml_params(xml_content, json_schema: dict | None = None): + """ + Compare the xml output to the json schema + + check if a value is a list - if so, get it's child elements + """ + root: Final = ET.fromstring(xml_content) + params: Final = {} + + if json_schema is not None: # check if we have a json schema for this function call + # iterate over all properties in the schema + for prop in json_schema["properties"]: + # If property is an array, get the nested items + _element = root.find(f"parameters/{prop}") + if json_schema["properties"][prop]["type"] == "array": + items = [] + if _element is not None: + for value in _element: + try: + if value.text is not None: + _value = json.loads(value.text) + else: + continue + except json.JSONDecodeError: + _value = value.text + items.append(_value) + params[prop] = items + # If property is not an array, append the value directly + elif _element is not None and _element.text is not None: + try: + _value = json.loads(_element.text) + except json.JSONDecodeError: + _value = _element.text + params[prop] = _value + else: + for child in root.findall(".//parameters/*"): + if child is not None and child.text is not None: + try: + # Attempt to decode the element's text as JSON + params[child.tag] = json.loads(child.text) + except json.JSONDecodeError: + # If JSON decoding fails, use the original text + params[child.tag] = child.text + + return params + + +### GEMINI HELPER FUNCTIONS ### + + +def get_system_prompt(messages): + system_prompt_indices: Final = [] + system_prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "system": + system_prompt += message["content"] + system_prompt_indices.append(idx) + if len(system_prompt_indices) > 0: + for idx in reversed(system_prompt_indices): + messages.pop(idx) + return system_prompt, messages + + +from litellm.types.llms.cohere import ( + CallObject, + ChatHistory, + ChatHistoryChatBot, + ChatHistorySystem, + ChatHistoryToolResult, + ChatHistoryUser, + ToolCallObject, + ToolResultObject, +) + + +def convert_openai_message_to_cohere_tool_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, + tool_calls: list, +) -> ToolResultObject: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "content": {"location": "San Francisco, CA", "unit": "fahrenheit", "temperature": "72"}, + }, + """ + """ + OpenAI message with a function call looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + + """ + Cohere tool_results look like: + { + "call": { + "name": "query_daily_sales_report", + "parameters": { + "day": "2023-09-29" + }, + }, + "outputs": [ + { + "date": "2023-09-29", + "summary": "Total Sales Amount: 10000, Total Units Sold: 250" + } + ] + }, + """ + + content_str: str = "" + if isinstance(message["content"], str): + content_str = message["content"] + elif isinstance(message["content"], list): + content_list: Final = message["content"] + for content in content_list: + if content["type"] == "text": + content_str += content["text"] + if len(content_str) > 0: + try: + content = json.loads(content_str) + except json.JSONDecodeError: + content = {"result": content_str} + else: + content = {} + name = "" + arguments = {} + # Recover name from last message with tool calls + if len(tool_calls) > 0: + tools: Final = tool_calls + msg_tool_call_id: Final = message.get("tool_call_id", None) + for tool in tools: + prev_tool_call_id = tool.get("id", None) + if msg_tool_call_id and prev_tool_call_id and msg_tool_call_id == prev_tool_call_id: + name = tool.get("function", {}).get("name", "") + arguments_str = tool.get("function", {}).get("arguments", "") + if arguments_str is not None and len(arguments_str) > 0: + arguments = json.loads(arguments_str) + + if message["role"] == "function": + function_message: Final[ChatCompletionFunctionMessage] = message + name = function_message["name"] + cohere_tool_result: ToolResultObject = { + "call": CallObject(name=name, parameters=arguments), + "outputs": [content], + } + return cohere_tool_result + else: + # We can't determine from openai message format whether it's a successful or + # error call result so default to the successful result template + + cohere_tool_result = { + "call": CallObject(name=name, parameters=arguments), + "outputs": [content], + } + return cohere_tool_result + + +def get_all_tool_calls(messages: list) -> list: + """ + Returns extracted list of `tool_calls`. + + Done to handle openai no longer returning tool call 'name' in tool results. + """ + tool_calls: Final[list] = [] + for m in messages: + if m.get("tool_calls", None) is not None: + if isinstance(m["tool_calls"], list): + tool_calls.extend(m["tool_calls"]) + + return tool_calls + + +def convert_to_cohere_tool_invoke(tool_calls: list) -> list[ToolCallObject]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + + """ + Cohere tool invokes: + { + "role": "CHATBOT", + "tool_calls": [{"name": "get_weather", "parameters": {"location": "San Francisco, CA"}}] + } + """ + + cohere_tool_invoke: Final[list[ToolCallObject]] = [ + { + "name": get_attribute_or_key(get_attribute_or_key(tool, "function"), "name"), + "parameters": json.loads(get_attribute_or_key(get_attribute_or_key(tool, "function"), "arguments")), + } + for tool in tool_calls + if get_attribute_or_key(tool, "type") == "function" + ] + + return cohere_tool_invoke + + +def cohere_messages_pt_v2( + messages: list, + model: str, + llm_provider: str, +) -> tuple[str | ToolResultObject, ChatHistory]: + """ + Returns a tuple(Union[tool_result, message], chat_history) + + - if last message is tool result -> return 'tool_result' + - if last message is text -> return message (str) + + - return preceding messages as 'chat_history' + + Note: + - cannot specify message if the last entry in chat history contains tool results + - message must be at least 1 token long or tool results must be specified. + - cannot specify tool_results if the last entry in chat history contains a user message + """ + tool_calls: Final[list] = get_all_tool_calls(messages=messages) + + ## GET MOST RECENT MESSAGE + most_recent_message: Final = messages.pop(-1) + returned_message: ToolResultObject | str = "" + if most_recent_message.get("role", "") is not None and most_recent_message["role"] == "tool": + # tool result + returned_message = convert_openai_message_to_cohere_tool_result(most_recent_message, tool_calls) + else: + content: Final[str | list] = most_recent_message.get("content") + if isinstance(content, str): + returned_message = content + else: + for chunk in content: + if chunk.get("type") == "text": + returned_message += chunk.get("text") + + ## CREATE CHAT HISTORY + user_message_types: Final = {"user"} + tool_message_types: Final = {"tool", "function"} + # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. + new_messages: Final[ChatHistory] = [] + msg_i = 0 + + while msg_i < len(messages): + user_content: str = "" + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] in user_message_types: + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + user_content += m["text"] + else: + user_content += messages[msg_i]["content"] + msg_i += 1 + + if len(user_content) > 0: + new_messages.append(ChatHistoryUser(role="USER", message=user_content)) + + system_content: str = "" + ## MERGE CONSECUTIVE SYSTEM CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + if isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + system_content += m["text"] + else: + system_content += messages[msg_i]["content"] + msg_i += 1 + + if len(system_content) > 0: + new_messages.append(ChatHistorySystem(role="SYSTEM", message=system_content)) + + assistant_content: str = "" + assistant_tool_calls: list[ToolCallObject] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + if messages[msg_i].get("content", None) is not None and isinstance(messages[msg_i]["content"], list): + for m in messages[msg_i]["content"]: + if m.get("type", "") == "text": + assistant_content += m["text"] + elif messages[msg_i].get("content") is not None and isinstance(messages[msg_i]["content"], str): + assistant_content += messages[msg_i]["content"] + if messages[msg_i].get("tool_calls", []): # support assistant tool invoke conversion + assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["tool_calls"])) + + if messages[msg_i].get("function_call"): + assistant_tool_calls.extend(convert_to_cohere_tool_invoke(messages[msg_i]["function_call"])) + + msg_i += 1 + + if len(assistant_content) > 0: + new_messages.append( + ChatHistoryChatBot( + role="CHATBOT", + message=assistant_content, + tool_calls=assistant_tool_calls, + ) + ) + + ## MERGE CONSECUTIVE TOOL RESULTS + tool_results: list[ToolResultObject] = [] + while msg_i < len(messages) and messages[msg_i]["role"] in tool_message_types: + tool_results.append(convert_openai_message_to_cohere_tool_result(messages[msg_i], tool_calls)) + + msg_i += 1 + + if len(tool_results) > 0: + new_messages.append(ChatHistoryToolResult(role="TOOL", tool_results=tool_results)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return returned_message, new_messages + + +def cohere_message_pt(messages: list): + tool_calls: Final[list] = get_all_tool_calls(messages=messages) + prompt = "" + tool_results: Final = [] + for message in messages: + # check if this is a tool_call result + if message["role"] == "tool": + tool_result = convert_openai_message_to_cohere_tool_result(message, tool_calls=tool_calls) + tool_results.append(tool_result) + elif message.get("content"): + prompt += message["content"] + "\n\n" + prompt = prompt.rstrip() + return prompt, tool_results + + +def amazon_titan_pt( + messages: list, +): # format - https://github.com/BerriAI/litellm/issues/1896 + """ + Amazon Titan uses 'User:' and 'Bot: in it's prompt template + """ + + class AmazonTitanConstants(Enum): + HUMAN_PROMPT = "\n\nUser: " # Assuming this is similar to Anthropic prompt formatting, since amazon titan's prompt formatting is currently undocumented + AI_PROMPT = "\n\nBot: " + + prompt = "" + for idx, message in enumerate(messages): + if message["role"] == "user": + prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" + elif message["role"] == "system": + prompt += f"{AmazonTitanConstants.HUMAN_PROMPT.value}{message['content']}" + else: + prompt += f"{AmazonTitanConstants.AI_PROMPT.value}{message['content']}" + if idx == 0 and message["role"] == "assistant": # ensure the prompt always starts with `\n\nHuman: ` + prompt = f"{AmazonTitanConstants.HUMAN_PROMPT.value}" + prompt + if messages[-1]["role"] != "assistant": + prompt += f"{AmazonTitanConstants.AI_PROMPT.value}" + return prompt + + +def _load_image_from_url(image_url): + try: + from PIL import Image + except Exception: + raise Exception("image conversion failed please run `pip install Pillow`") + from io import BytesIO + + try: + # Send a GET request to the image URL + client: Final = HTTPHandler(concurrent_limit=1) + response: Final[httpx.Response] = safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + # Check the response's content type to ensure it is an image + content_type: Final = response.headers.get("content-type") + if not content_type or "image" not in content_type: + raise ValueError(f"URL does not point to a valid image (content-type: {content_type})") + + # Load the image from the response content + return Image.open(BytesIO(response.content)) + + except Exception as e: + raise e + + +def _gemini_vision_convert_messages(messages: list): + """ + Converts given messages for GPT-4 Vision to Gemini format. + + Args: + messages (list): The messages to convert. Each message can be a dictionary with a "content" key. The content can be a string or a list of elements. If it is a string, it will be concatenated to the prompt. If it is a list, each element will be processed based on its type: + - If the element is a dictionary with a "type" key equal to "text", its "text" value will be concatenated to the prompt. + - If the element is a dictionary with a "type" key equal to "image_url", its "image_url" value will be added to the list of images. + + Returns: + tuple: A tuple containing the prompt (a string) and the processed images (a list of objects representing the images). + """ + + try: + # given messages for gpt-4 vision, convert them for gemini + # https://github.com/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_python.ipynb + prompt = "" + images: Final = [] + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + elif element["type"] == "image_url": + image_url = element["image_url"]["url"] + images.append(image_url) + # processing images passed to gemini + processed_images: Final = [] + for img in images: + if "https:/" in img: + # Case 1: Image from URL + image = _load_image_from_url(img) + processed_images.append(image) + + else: + try: + from PIL import Image + except Exception: + raise Exception("gemini image conversion failed please run `pip install Pillow`") + + if "base64" in img: + # Case 2: Base64 image data + import base64 + import io + + # Extract the base64 image data + base64_data = img.split("base64,")[1] + + # Decode the base64 image data + image_data = base64.b64decode(base64_data) + + # Load the image from the decoded data + image = Image.open(io.BytesIO(image_data)) + else: + # Case 3: Image filepath (e.g. temp.jpeg) given + image = Image.open(img) + processed_images.append(image) + content: Final = [prompt] + processed_images + return content + except Exception as e: + raise e + + +def gemini_text_image_pt(messages: list): + """ + { + "contents":[ + { + "parts":[ + {"text": "What is this picture?"}, + { + "inline_data": { + "mime_type":"image/jpeg", + "data": "'$(base64 -w0 image.jpg)'" + } + } + ] + } + ] + } + """ + try: + pass + except Exception: + raise Exception("Importing google.generativeai failed, please run 'pip install -q google-generativeai") + + prompt = "" + images: Final = [] + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + elif element["type"] == "image_url": + image_url = element["image_url"]["url"] + images.append(image_url) + + content: Final = [prompt] + images + return content + + +def azure_text_pt(messages: list): + prompt = "" + for message in messages: + if isinstance(message["content"], str): + prompt += message["content"] + elif isinstance(message["content"], list): + # see https://docs.litellm.ai/docs/providers/openai#openai-vision-models + for element in message["content"]: + if isinstance(element, dict): + if element["type"] == "text": + prompt += element["text"] + return prompt + + +###### AZURE AI ####### +def stringify_json_tool_call_content(messages: list) -> list: + """ + + - Check 'content' in tool role -> convert to dict (if not) -> stringify + + Done for azure_ai/cohere calls to handle results of a tool call + """ + + for m in messages: + if m["role"] == "tool" and isinstance(m["content"], str): + # check if content is a valid json object + try: + json.loads(m["content"]) + except json.JSONDecodeError: + m["content"] = json.dumps({"result": m["content"]}) + + return messages + + +###### AMAZON BEDROCK ####### + +from email.message import Message + +import httpx + +from litellm.types.llms.bedrock import ( + BedrockConverseReasoningContentBlock, + BedrockConverseReasoningTextBlock, + BedrockToolSpec, + SearchResultBlock, +) +from litellm.types.llms.bedrock import ContentBlock as BedrockContentBlock +from litellm.types.llms.bedrock import DocumentBlock as BedrockDocumentBlock +from litellm.types.llms.bedrock import ImageBlock as BedrockImageBlock +from litellm.types.llms.bedrock import SourceBlock as BedrockSourceBlock +from litellm.types.llms.bedrock import ToolBlock as BedrockToolBlock +from litellm.types.llms.bedrock import ToolResultBlock as BedrockToolResultBlock +from litellm.types.llms.bedrock import ( + ToolResultContentBlock as BedrockToolResultContentBlock, +) +from litellm.types.llms.bedrock import ToolUseBlock as BedrockToolUseBlock +from litellm.types.llms.bedrock import VideoBlock as BedrockVideoBlock + + +def _parse_content_type(content_type: str) -> str: + m: Final = Message() + m["content-type"] = content_type + return m.get_content_type() + + +def _parse_mime_type(base64_data: str) -> str | None: + mime_type_match: Final = re.match(r"data:(.*?);base64", base64_data) + if mime_type_match: + return mime_type_match.group(1) + else: + return None + + +class BedrockImageProcessor: + """Handles both sync and async image processing for Bedrock conversations.""" + + @staticmethod + def _post_call_image_processing(response: httpx.Response, image_url: str = "") -> tuple[str, str]: + # Check the response's content type to ensure it is an image + content_type: str | None = response.headers.get("content-type") + + # Use helper function to infer content type with fallback logic + content_type = infer_content_type_from_url_and_content( + url=image_url, + content=response.content, + current_content_type=content_type, + ) + + content_type = _parse_content_type(content_type) + + # Convert the image content to base64 bytes + base64_bytes: Final = base64.b64encode(response.content).decode("utf-8") + + return base64_bytes, content_type + + @staticmethod + async def get_image_details_async(image_url) -> tuple[str, str]: + try: + client: Final = get_async_httpx_client( + llm_provider=httpxSpecialProvider.PromptFactory, + params={"concurrent_limit": 1}, + ) + # Send a GET request to the image URL + response: Final[httpx.Response] = await async_safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + return BedrockImageProcessor._post_call_image_processing(response, image_url) + + except Exception as e: + raise e + + @staticmethod + def get_image_details(image_url) -> tuple[str, str]: + try: + client: Final = HTTPHandler(concurrent_limit=1) + # Send a GET request to the image URL + response: Final[httpx.Response] = safe_get(client, image_url) + response.raise_for_status() # Raise an exception for HTTP errors + + return BedrockImageProcessor._post_call_image_processing(response, image_url) + + except Exception as e: + raise e + + @staticmethod + def _parse_base64_image(image_url: str) -> tuple[str, str, str]: + """Parse base64 encoded image data.""" + image_metadata, img_without_base_64 = image_url.split(",") + + # Extract MIME type using regular expression + mime_type_match: Final = re.match(r"data:(.*?);base64", image_metadata) + + if mime_type_match: + mime_type = mime_type_match.group(1) + mime_type = mime_type.split(";")[0] + image_format = mime_type.split("/")[1] + else: + mime_type = "image/jpeg" + image_format = "jpeg" + + return img_without_base_64, mime_type, image_format + + @staticmethod + def _validate_format(mime_type: str, image_format: str) -> str: + """Validate image format and mime type for both images and documents.""" + + supported_image_formats: Final = litellm.AmazonConverseConfig().get_supported_image_types() + supported_doc_formats: Final = litellm.AmazonConverseConfig().get_supported_document_types() + supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() + + document_types: Final = ["application", "text"] + is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) + + supported_image_and_video_formats: Final[list[str]] = supported_video_formats + supported_image_formats + + if is_document: + return BedrockImageProcessor._get_document_format( + mime_type=mime_type, supported_doc_formats=supported_doc_formats + ) + + else: + ######################################################### + # Check if image_format is an image or video + ######################################################### + if image_format not in supported_image_and_video_formats: + raise ValueError( + f"Unsupported image format: {image_format}. Supported formats: {supported_image_and_video_formats}" + ) + return image_format + + @staticmethod + def _get_document_format(mime_type: str, supported_doc_formats: list[str]) -> str: + """ + Get the document format from the mime type + + - Primary method - uses `mimetypes.guess_all_extensions` + - Fallback method - uses `get_file_extension_from_mime_type` + + Relevant Issue: https://github.com/BerriAI/litellm/issues/12260 + + `mimetypes` is not available in docker containers, so we fallback to `get_file_extension_from_mime_type` + + Args: + mime_type: The mime type of the document + supported_doc_formats: The supported document formats for the current model + + Returns: + The document format + """ + valid_extensions: list[str] | None = None + potential_extensions: Final = mimetypes.guess_all_extensions(mime_type, strict=False) + valid_extensions = [ext[1:] for ext in potential_extensions if ext[1:] in supported_doc_formats] + + # Fallback to types/files.py if mimetypes doesn't return valid extensions + ################# + # litellm runs on docker containers and `mimetypes` depends on the installed mimetypes of the OS + # we fallback to well known mime types in types/files.py if mimetypes doesn't return valid extensions + if not valid_extensions: + try: + fallback_extension: Final = get_file_extension_from_mime_type(mime_type) + if fallback_extension in supported_doc_formats: + valid_extensions = [fallback_extension] + except ValueError: + # Neither mimetypes nor files.py could handle this MIME type + # get_file_extension_from_mime_type raises ValueError if the mime type is not supported + pass + + if not valid_extensions: + raise ValueError( + f"No supported extensions for MIME type: {mime_type}. Supported formats: {supported_doc_formats}" + ) + + # Use first valid extension instead of provided image_format + return valid_extensions[0] + + @staticmethod + def _create_bedrock_block(image_bytes: str, mime_type: str, image_format: str) -> BedrockContentBlock: + """Create appropriate Bedrock content block based on mime type.""" + _blob: Final = BedrockSourceBlock(bytes=image_bytes) + + document_types: Final = ["application", "text"] + is_document: Final = any(mime_type.startswith(doc_type) for doc_type in document_types) + + supported_video_formats: Final = litellm.AmazonConverseConfig().get_supported_video_types() + is_video: Final = any(image_format.startswith(video_type) for video_type in supported_video_formats) + + HASH_SAMPLE_BYTES: Final = 64 * 1024 # hash up to 64 KB of data + + if is_document: + # --- Prepare normalized bytes for hashing (without modifying original) --- + if isinstance(image_bytes, str): + # Remove whitespace/newlines so base64 variations hash identically + normalized = "".join(image_bytes.split()).encode("utf-8") + else: + normalized = image_bytes + + # --- Use only the first 64 KB for speed --- + if len(normalized) <= HASH_SAMPLE_BYTES: + sample = normalized + else: + sample = normalized[:HASH_SAMPLE_BYTES] + + # --- Compute deterministic hash (sample + total length) --- + hasher: Final = hashlib.sha256() + hasher.update(sample) + hasher.update(str(len(normalized)).encode("utf-8")) # include full length for uniqueness + full_hash: Final = hasher.hexdigest() + content_hash: Final = full_hash[:16] # short deterministic ID + + document_name: Final = f"DocumentPDFmessages_{content_hash}_{image_format}" + + # --- Return content block --- + return BedrockContentBlock( + document=BedrockDocumentBlock( + source=_blob, + format=image_format, + name=document_name, + ) + ) + elif is_video: + return BedrockContentBlock(video=BedrockVideoBlock(source=_blob, format=image_format)) + else: + return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) + + @classmethod + def process_image_sync(cls, image_url: str, format: str | None = None) -> BedrockContentBlock: + """Synchronous image processing.""" + + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "http://" in image_url or "https://" in image_url: + img_bytes, mime_type = BedrockImageProcessor.get_image_details(image_url) + image_format = mime_type.split("/")[1] + else: + raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") + + if format: + mime_type = format + image_format = mime_type.split("/")[1] + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) + + @classmethod + async def process_image_async(cls, image_url: str, format: str | None) -> BedrockContentBlock: + """Asynchronous image processing.""" + + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "http://" in image_url or "https://" in image_url: + img_bytes, mime_type = await BedrockImageProcessor.get_image_details_async(image_url) + image_format = mime_type.split("/")[1] + else: + raise ValueError("Unsupported image type. Expected either image url or base64 encoded string") + + if format: # override with user-defined params + mime_type = format + image_format = mime_type.split("/")[1] + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) + + +def _convert_to_bedrock_tool_call_invoke( + tool_calls: list, + model: str | None = None, +) -> list[BedrockContentBlock]: + """ + OpenAI tool invokes: + { + "role": "assistant", + "content": null, + "tool_calls": [ + { + "id": "call_abc123", + "type": "function", + "function": { + "name": "get_current_weather", + "arguments": "{\n\"location\": \"Boston, MA\"\n}" + } + } + ] + }, + """ + """ + Bedrock tool invokes: + [ + { + "role": "assistant", + "toolUse": { + "input": {"location": "Boston, MA", ..}, + "name": "get_current_weather", + "toolUseId": "call_abc123" + } + } + ] + """ + """ + - json.loads argument + - extract name + - extract id + """ + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + split_concatenated_json_objects, + ) + + try: + _parts_list: Final[list[BedrockContentBlock]] = [] + for tool in tool_calls: + if "function" in tool: + tool_id = tool["id"] + name = make_valid_bedrock_tool_name(tool["function"].get("name", "")) + arguments = tool["function"].get("arguments", "") + + if not arguments or not arguments.strip(): + arguments_dict = {} + else: + try: + arguments_dict = json.loads(arguments) + # Ensure arguments_dict is always a dict + # (Bedrock requires toolUse.input to be an object). + # Some providers return arguments: '""' which + # json.loads decodes to a bare string. + if not isinstance(arguments_dict, dict): + arguments_dict = {} + except json.JSONDecodeError: + # The model may return multiple JSON objects + # concatenated in a single arguments string, e.g. + # '{"cmd":"a"}{"cmd":"b"}{"cmd":"c"}' + # Split them and emit one toolUse block per object. + # Fixes: https://github.com/BerriAI/litellm/issues/20543 + parsed_objects = split_concatenated_json_objects(arguments) + if parsed_objects: + # First object keeps the original tool id. + for obj_idx, obj in enumerate(parsed_objects): + block_id = _sanitize_bedrock_tool_use_id( + tool_id if obj_idx == 0 else f"{tool_id}_{obj_idx}" + ) + bedrock_tool = BedrockToolUseBlock(input=obj, name=name, toolUseId=block_id) + _parts_list.append(BedrockContentBlock(toolUse=bedrock_tool)) + # cache_control applies to the whole original + # tool call; attach after the last split block. + if tool.get("cache_control", None) is not None: + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool["cache_control"]}, + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts_list.append(_cache_point_block) + continue + # Fallback: no objects extracted — use empty dict. + arguments_dict = {} + + bedrock_tool = BedrockToolUseBlock( + input=arguments_dict, name=name, toolUseId=_sanitize_bedrock_tool_use_id(tool_id) + ) + bedrock_content_block = BedrockContentBlock(toolUse=bedrock_tool) + _parts_list.append(bedrock_content_block) + + # Check for cache_control and add a separate cachePoint block + if tool.get("cache_control", None) is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool["cache_control"]}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + _parts_list.append(cache_point_block) + return _parts_list + except Exception as e: + tool_call_ids: Final = tuple(tool.get("id") for tool in tool_calls if isinstance(tool, dict)) + raise litellm.BadRequestError( + message=f"Unable to convert openai tool calls with ids={tool_call_ids} to bedrock tool calls. " + f"Received error={e}", + model=model or "", + llm_provider="bedrock", + ) from e + + +def _append_bedrock_tool_result_media_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + processed_block: BedrockContentBlock, + content: dict, + content_type: str, +) -> None: + if "image" in processed_block: + tool_result_content_blocks.append(BedrockToolResultContentBlock(image=processed_block["image"])) + elif "document" in processed_block: + tool_result_content_blocks.append(BedrockToolResultContentBlock(document=processed_block["document"])) + else: + verbose_logger.warning( + "Bedrock Converse: unrecognized BedrockContentBlock keys %s for %s tool-result block %s; dropping.", + list(processed_block.keys()), + content_type, + content, + ) + + +def _append_bedrock_tool_result_image_url_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + content: dict, +) -> None: + format: str | None = None + if isinstance(content["image_url"], dict): + image_url = content["image_url"]["url"] + format = content["image_url"].get("format") + else: + image_url = content["image_url"] + processed_block: Final = BedrockImageProcessor.process_image_sync( + image_url=image_url, + format=format, + ) + _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "image_url") + + +def _append_bedrock_tool_result_file_block( + tool_result_content_blocks: list[BedrockToolResultContentBlock], + content: dict, +) -> None: + # Match the user-message path (_process_file_message): accept either + # file_data (base64 data URI) or file_id (server-side reference / URL). + file_obj: Final = content.get("file") or {} + file_data: Final = file_obj.get("file_data") + file_id: Final = file_obj.get("file_id") + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={content}", + model="", + llm_provider="bedrock", + ) + processed_block: Final = BedrockImageProcessor.process_image_sync( + image_url=cast(str, file_id or file_data), + format=file_obj.get("format"), + ) + _append_bedrock_tool_result_media_block(tool_result_content_blocks, processed_block, content, "file") + + +def _parse_bedrock_tool_result_content_list( + content_list: list, +) -> list[BedrockToolResultContentBlock]: + tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] + for content in content_list: + if content["type"] == "text": + tool_result_content_blocks.append(BedrockToolResultContentBlock(text=content["text"])) + elif content["type"] == "image_url": + _append_bedrock_tool_result_image_url_block(tool_result_content_blocks, content) + elif content["type"] == "file": + _append_bedrock_tool_result_file_block(tool_result_content_blocks, content) + return tool_result_content_blocks + + +def _build_bedrock_tool_result_content_blocks( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, +) -> tuple[list[BedrockToolResultContentBlock], bool]: + # Optional OpenAI tool-message extension: + # allow structured Bedrock search results on tool messages and map them + # directly to toolResult.content[].searchResult for Converse API. + # + # If `search_results` is present, we intentionally prefer it over `content` + # to avoid generating mixed text + searchResult blocks. + search_results: Final = message.get("search_results") + if isinstance(search_results, list): + tool_result_content_blocks: Final[list[BedrockToolResultContentBlock]] = [] + for result in search_results: + if not isinstance(result, dict): + continue + tool_result_content_blocks.append( + BedrockToolResultContentBlock(searchResult=cast(SearchResultBlock, result)) + ) + if tool_result_content_blocks: + return tool_result_content_blocks, True + + message_content: Final = message["content"] + if isinstance(message_content, str): + return [BedrockToolResultContentBlock(text=message_content)], False + if isinstance(message_content, list): + return _parse_bedrock_tool_result_content_list(message_content), False + return [], False + + +def _convert_to_bedrock_tool_call_result( + message: ChatCompletionToolMessage | ChatCompletionFunctionMessage, +) -> BedrockContentBlock: + """ + OpenAI message with a tool result looks like: + { + "tool_call_id": "tool_1", + "role": "tool", + "name": "get_current_weather", + "content": "function result goes here", + }, + + OpenAI message with a function call result looks like: + { + "role": "function", + "name": "get_current_weather", + "content": "function result goes here", + } + """ + """ + Bedrock result looks like this: + { + "role": "user", + "content": [ + { + "toolResult": { + "toolUseId": "tooluse_kZJMlvQmRJ6eAyJE5GIl7Q", + "content": [ + { + "json": { + "song": "Elemental Hotel", + "artist": "8 Storey Hike" + } + } + ] + } + } + ] + } + """ + """ + - + """ + tool_result_content_blocks, used_search_results = _build_bedrock_tool_result_content_blocks(message) + + message.get("name", "") + id: Final = _sanitize_bedrock_tool_use_id(str(message.get("tool_call_id", str(uuid.uuid4())))) + + tool_result: Final = BedrockToolResultBlock(content=tool_result_content_blocks, toolUseId=id) + if used_search_results: + tool_result["status"] = cast(Literal["success"], "success") + + content_block: Final = BedrockContentBlock(toolResult=tool_result) + + return content_block + + +def _deduplicate_bedrock_content_blocks( + blocks: list[BedrockContentBlock], + block_key: str, + id_key: str = "toolUseId", +) -> list[BedrockContentBlock]: + """ + Remove duplicate content blocks that share the same ID under ``block_key``. + + Bedrock requires all toolResult and toolUse IDs within a single message to + be unique. When merging consecutive messages, duplicates can occur if the + same tool_call_id appears multiple times in conversation history. + + When duplicates exist, the first occurrence is retained and subsequent ones + are discarded. A warning is logged for every dropped block so that + upstream duplication bugs remain visible. + + Blocks that do not contain ``block_key`` (e.g., cachePoint, text) are + always preserved. + + Args: + blocks: The list of Bedrock content blocks to deduplicate. + block_key: The dict key to inspect (e.g. ``"toolResult"`` or ``"toolUse"``). + id_key: The nested key that holds the unique ID (default ``"toolUseId"``). + """ + seen_ids: Final[set[str]] = set() + deduplicated: Final[list[BedrockContentBlock]] = [] + for block in blocks: + keyed = block.get(block_key) + if keyed is not None and isinstance(keyed, dict): + block_id = keyed.get(id_key) + if block_id: + if block_id in seen_ids: + verbose_logger.warning( + "Bedrock Converse: dropping duplicate %s block with " + "%s=%s. This may indicate duplicate tool messages in " + "conversation history.", + block_key, + id_key, + block_id, + ) + continue + seen_ids.add(block_id) + deduplicated.append(block) + return deduplicated + + +def _deduplicate_bedrock_tool_content( + tool_content: list[BedrockContentBlock], +) -> list[BedrockContentBlock]: + """Convenience wrapper: deduplicate ``toolResult`` blocks by ``toolUseId``.""" + return _deduplicate_bedrock_content_blocks(tool_content, "toolResult") + + +def _rename_duplicate_bedrock_document_names( + contents: list[BedrockMessageBlock], +) -> list[BedrockMessageBlock]: + """ + Rename duplicate document names across all messages in a Bedrock request. + + Document names are derived from a content hash, so the same file appearing + in multiple conversation turns produces identical names and Bedrock rejects + the request with "Messages can not contain duplicate document names". The + first occurrence keeps its original name so prompt-cache prefixes stay + stable; later occurrences get a deterministic positional suffix + (``_2``, ``_3``, ...), bumped further if the suffixed name already + belongs to another document (e.g. an organic name ending in ``_2``). + """ + used_names: Final[set[str]] = set() + for message in contents: + for block in message.get("content") or []: + document = block.get("document") + if isinstance(document, dict) and document.get("name"): + used_names.add(document["name"]) + + name_counts: Final[dict[str, int]] = {} + for message in contents: + for block in message.get("content") or []: + document = block.get("document") + if not isinstance(document, dict): + continue + name = document.get("name") + if not name: + continue + count = name_counts.get(name, 0) + 1 + name_counts[name] = count + if count > 1: + suffix = count + new_name = f"{name}_{suffix}" + while new_name in used_names: + suffix += 1 + new_name = f"{name}_{suffix}" + used_names.add(new_name) + document["name"] = new_name + return contents + + +BEDROCK_DOCUMENT_PLACEHOLDER_TEXT: Final = "." + + +def _with_text_when_document_only(message: BedrockMessageBlock) -> BedrockMessageBlock: + blocks: Final = message["content"] + needs_text: Final = ( + message["role"] == "user" + and any("document" in block for block in blocks) + and all("text" not in block for block in blocks) + ) + if not needs_text: + return message + placeholder: Final = BedrockContentBlock(text=BEDROCK_DOCUMENT_PLACEHOLDER_TEXT) + cut: Final = len(blocks) - 1 if "cachePoint" in blocks[-1] else len(blocks) + return BedrockMessageBlock(role="user", content=[*blocks[:cut], placeholder, *blocks[cut:]]) + + +def _ensure_document_messages_have_text( + contents: list[BedrockMessageBlock], +) -> list[BedrockMessageBlock]: + """ + Bedrock Converse rejects any user message that carries a document block + without a sibling text block ("A text block must be included when using + documents"), e.g. Claude Code sends the PDF as a document-only user turn. + Inject a placeholder text block, kept ahead of a trailing cachePoint so + the caller's cache boundary stays the final block. + """ + return [_with_text_when_document_only(message) for message in contents] + + +def _sort_bedrock_assistant_content_blocks( + blocks: list[BedrockContentBlock], +) -> list[BedrockContentBlock]: + """ + Sort assistant content blocks so that ``text`` blocks appear before + ``toolUse`` blocks. + + Bedrock requires all ``text`` blocks to precede any ``toolUse`` blocks + within an assistant message. When the Responses API converts + function_call items before message items, the resulting ``toolUse`` + blocks can end up before ``text`` blocks, causing Bedrock to reject + the request with a 400 error because the ``toolUse`` → ``toolResult`` + pairing is broken by the intervening ``text`` block. + + Sort order (stable) inside each cache-point segment: + 0 - reasoningContent + 1 - text / image / document / video / other non-tool blocks + 2 - toolUse + 3 - cachePoint (kept at its original boundary) + """ + + cache_point_indices: Final[tuple[int, ...]] = tuple( + index for index, block in enumerate(blocks) if "cachePoint" in block + ) + + def _sort_key(indexed_block: tuple[int, BedrockContentBlock]) -> tuple[int, int, int]: + index, block = indexed_block + segment_index: Final[int] = bisect_left(cache_point_indices, index) + if "reasoningContent" in block: + return segment_index, 0, index + if "cachePoint" in block: + return segment_index, 3, index + if "toolUse" in block: + return segment_index, 2, index + return segment_index, 1, index + + sorted_indexed_blocks: Final = sorted(enumerate(blocks), key=_sort_key) + return [ # mutable-ok: Bedrock Converse requires a list of content blocks + block for _, block in sorted_indexed_blocks + ] + + +def _insert_assistant_continue_message( + messages: list[BedrockMessageBlock], + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> list[BedrockMessageBlock]: + """ + Add dummy message between user/tool result blocks. + + Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 + """ + if assistant_continue_message is not None: + if isinstance(assistant_continue_message, str): + messages.append( + BedrockMessageBlock( + role="assistant", + content=[BedrockContentBlock(text=assistant_continue_message)], + ) + ) + elif isinstance(assistant_continue_message, dict): + text = convert_content_list_to_str(assistant_continue_message) + messages.append( + BedrockMessageBlock( + role="assistant", + content=[BedrockContentBlock(text=text)], + ) + ) + elif litellm.modify_params: + text = convert_content_list_to_str(cast(ChatCompletionAssistantMessage, DEFAULT_ASSISTANT_CONTINUE_MESSAGE)) + messages.append( + BedrockMessageBlock( + role="assistant", + content=[ + BedrockContentBlock(text=text), + ], + ) + ) + return messages + + +def get_user_message_block_or_continue_message( + message: ChatCompletionUserMessage, + user_continue_message: ChatCompletionUserMessage | None = None, +) -> ChatCompletionUserMessage: + """ + Returns the user content block + if content block is an empty string, then return the default continue message + + Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 + """ + content_block: Final = message.get("content", None) + + # Handle None case + if content_block is None or (user_continue_message is None and litellm.modify_params is False): + return skip_empty_text_blocks(message=message) + + # Handle string case + if isinstance(content_block, str): + # check if content is empty + if content_block.strip(): + return message + else: + return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) + + # Handle list case + if isinstance(content_block, list): + """ + CHECK FOR + "content": [ + { + "type": "text", + "text": "" + } + ], + """ + if not content_block: + return ChatCompletionUserMessage(**(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE)) + # Create a copy of the message to avoid modifying the original + modified_content_block: Final = content_block.copy() + + for item in modified_content_block: + # Check if the list is empty + if item["type"] == "text": + if not item["text"].strip(): + # Replace empty text with continue message + _user_continue_message = ChatCompletionUserMessage( + **(user_continue_message or DEFAULT_USER_CONTINUE_MESSAGE) + ) + text = convert_content_list_to_str(_user_continue_message) + item["text"] = text + break + modified_message: Final = message.copy() + modified_message["content"] = modified_content_block + return modified_message + + # Handle unsupported type + raise ValueError(f"Unsupported content type: {type(content_block)}") + + +def return_assistant_continue_message( + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + if assistant_continue_message and isinstance(assistant_continue_message, str): + return ChatCompletionAssistantMessage( + role="assistant", + content=assistant_continue_message, + ) + elif assistant_continue_message and isinstance(assistant_continue_message, dict): + return ChatCompletionAssistantMessage(**assistant_continue_message) + else: + return DEFAULT_ASSISTANT_CONTINUE_MESSAGE + + +def _skip_empty_dict_blocks(blocks: list[dict]) -> list[dict]: + """ + Filter out empty text blocks from a list of dictionaries. + + Args: + blocks: List of dictionaries representing message content blocks + + Returns: + Filtered list of non-empty text blocks + """ + return [item for item in blocks if not (item.get("type") == "text" and not item.get("text", "").strip())] + + +@overload +def skip_empty_text_blocks( + message: ChatCompletionAssistantMessage, +) -> ChatCompletionAssistantMessage: + pass + + +@overload +def skip_empty_text_blocks( + message: ChatCompletionUserMessage, +) -> ChatCompletionUserMessage: + pass + + +def skip_empty_text_blocks( + message: ChatCompletionAssistantMessage | ChatCompletionUserMessage, +) -> ChatCompletionAssistantMessage | ChatCompletionUserMessage: + """ + Skips empty text blocks in message content text blocks. + + Do not insert content here. This is a helper function, which can also be used in base case. + """ + content_block: Final = message.get("content", None) + if content_block is None: + return message + if ( + isinstance(content_block, str) + and not content_block.strip() + and is_non_content_values_set(message) + and message["role"] == "assistant" + ): + modified_message = message.copy() + modified_message["content"] = None # user message content cannot be None + return modified_message + elif isinstance(content_block, list): + modified_content_block: Final = _skip_empty_dict_blocks(cast(list[dict], content_block)) + + # If no content remains and it's an assistant message, set content to None + if not modified_content_block and message["role"] == "assistant": + modified_message = message.copy() + modified_message["content"] = None + return modified_message + + modified_message_alt: Final = message.copy() + + # Type-specific casting based on message role + if message["role"] == "assistant": + modified_message_alt["content"] = cast( + list[OpenAIMessageContentListBlock] | None, + modified_content_block or None, + ) + elif message["role"] == "user" and modified_content_block is not None: + modified_message_alt["content"] = cast(list[ChatCompletionTextObject] | None, modified_content_block) + + return modified_message_alt + + return message + + +def process_empty_text_blocks( + message: ChatCompletionAssistantMessage, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + modified_content_block = message.get("content", None) + ## BASE CASE ## + if modified_content_block is None or not isinstance(modified_content_block, list): + return message + + # Check if all items are empty text blocks + if all(item["type"] == "text" and not item["text"].strip() for item in modified_content_block): + # Replace with a single continue message + _assistant_continue_message: Final = return_assistant_continue_message(assistant_continue_message) + modified_content_block = [ + { + "type": "text", + "text": convert_content_list_to_str(_assistant_continue_message), + } + ] + else: + # Filter out only empty text blocks, keeping non-empty text and other block types + modified_content_block = [ + item for item in modified_content_block if not (item["type"] == "text" and not item["text"].strip()) + ] + + modified_message: Final = message.copy() + modified_message["content"] = cast( + list[ChatCompletionTextObject] | list[ChatCompletionThinkingBlock], + modified_content_block, + ) + return modified_message + + +def get_assistant_message_block_or_continue_message( + message: ChatCompletionAssistantMessage, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> ChatCompletionAssistantMessage: + """ + Returns the user content block + if content block is an empty string, then return the default continue message + + Relevant Issue: https://github.com/BerriAI/litellm/issues/7169 + """ + content_block: Final = message.get("content", None) + + # Handle Base case + if content_block is None or (assistant_continue_message is None and litellm.modify_params is False): + return skip_empty_text_blocks(message=message) + + # Handle string case + if isinstance(content_block, str): + # check if content is empty + if content_block.strip(): + return message + else: + if is_non_content_values_set(message): + modified_message: Final = message.copy() + modified_message["content"] = None + return modified_message + return return_assistant_continue_message(assistant_continue_message) + + # Handle list case + if isinstance(content_block, list): + """ + CHECK FOR + "content": [ + { + "type": "text", + "text": "" + } + ], + """ + return process_empty_text_blocks(message=message, assistant_continue_message=assistant_continue_message) + + # Handle unsupported type + raise ValueError(f"Unsupported content type: {type(content_block)}") + + +class BedrockConverseMessagesProcessor: + @staticmethod + def _initial_message_setup( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + ) -> list: + # gracefully handle base case of no messages at all + if len(messages) == 0: + if user_continue_message is not None: + messages.append(user_continue_message) + elif litellm.modify_params: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE) + else: + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + "bedrock requires at least one non-system message", + model=model, + llm_provider=llm_provider, + ) + + # if initial message is assistant message + if messages[0].get("role") is not None and messages[0]["role"] == "assistant": + if not messages[0].get("prefix"): + if user_continue_message is not None: + messages.insert(0, user_continue_message) + elif litellm.modify_params: + messages.insert(0, DEFAULT_USER_CONTINUE_MESSAGE) + + # if final message is assistant message + if messages[-1].get("role") is not None and messages[-1]["role"] == "assistant": + if not messages[-1].get("prefix"): + if user_continue_message is not None: + messages.append(user_continue_message) + elif litellm.modify_params: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE) + return messages + + @staticmethod + async def _bedrock_converse_messages_pt_async( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, + ) -> list[BedrockMessageBlock]: + contents: list[BedrockMessageBlock] = [] + msg_i = 0 + + messages = BedrockConverseMessagesProcessor._initial_message_setup( + messages, model, llm_provider, user_continue_message + ) + + while msg_i < len(messages): + user_content: list[BedrockContentBlock] = [] + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "user": + message_block = get_user_message_block_or_continue_message( + message=messages[msg_i], + user_continue_message=user_continue_message, + ) + if isinstance(message_block["content"], list): + _parts: list[BedrockContentBlock] = [] + for element in message_block["content"]: + if isinstance(element, dict): + if element["type"] == "text": + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "guarded_text": + # Wrap guarded_text in guardContent block + _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) + _parts.append(_part) + elif element["type"] in ("grounding_source", "query"): + # Contextual grounding tags are guardrail metadata; the + # model only needs the underlying text, so render them + # as plain text on the generate path. + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "image_url": + format: str | None = None + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + format = element["image_url"].get("format") + else: + image_url = element["image_url"] + _part = await BedrockImageProcessor.process_image_async( + image_url=image_url, format=format + ) + _parts.append(_part) + elif element["type"] == "file": + _part = await BedrockConverseMessagesProcessor._async_process_file_message( + message=cast(ChatCompletionFileObject, element) + ) + _parts.append(_part) + elif element["type"] == "document": + _part = BedrockConverseMessagesProcessor._process_document_message(element) + _parts.append(_part) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + user_content.extend(_parts) + elif message_block["content"] and isinstance(message_block["content"], str): + _part = BedrockContentBlock(text=messages[msg_i]["content"]) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + user_content.append(_part) + if _cache_point_block is not None: + user_content.append(_cache_point_block) + + msg_i += 1 + if user_content: + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=user_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(user_content) + else: + contents.append(BedrockMessageBlock(role="user", content=user_content)) + + ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## + tool_content: list[BedrockContentBlock] = [] + while msg_i < len(messages) and messages[msg_i]["role"] == "tool": + current_message = messages[msg_i] + tool_call_result = _convert_to_bedrock_tool_call_result(current_message) + tool_content.append(tool_call_result) + + # Check if we need to add a separate cachePoint block + tool_msg_cache_control = None + + # Check for message-level cache_control + if current_message.get("cache_control", None) is not None: + tool_msg_cache_control = current_message["cache_control"] + # Check for content-level cache_control in list content + elif isinstance(current_message.get("content"), list): + for content_element in current_message["content"]: + if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: + tool_msg_cache_control = content_element["cache_control"] + break + + # Add a separate cachePoint block if cache_control is present + if tool_msg_cache_control is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool_msg_cache_control}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + tool_content.append(cache_point_block) + + msg_i += 1 + # Deduplicate toolResult blocks with the same toolUseId + tool_content = _deduplicate_bedrock_tool_content(tool_content) + if tool_content: + # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(tool_content) + else: + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + assistant_content: list[BedrockContentBlock] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_message_block = get_assistant_message_block_or_continue_message( + message=messages[msg_i], + assistant_continue_message=assistant_continue_message, + ) + assistant_message_content_start: int = len(assistant_content) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + _assistant_content = assistant_message_block.get("content", None) + thinking_blocks = cast( + list[ChatCompletionThinkingBlock] | None, + assistant_message_block.get("thinking_blocks"), + ) + + if thinking_blocks is not None: + converted_thinking_blocks = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, + ) + + if _assistant_content is not None and isinstance(_assistant_content, list): + assistants_parts: list[BedrockContentBlock] = [] + for element in _assistant_content: + if isinstance(element, dict): + if element["type"] == "thinking": + thinking_block = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] + ) + assistants_parts = ( + BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) + ) + elif element["type"] == "text": + # Skip completely empty strings to avoid blank content blocks + if element.get("text", "").strip(): + assistants_part = BedrockContentBlock(text=element["text"]) + assistants_parts.append(assistants_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + assistants_part = await BedrockImageProcessor.process_image_async(image_url=image_url) + assistants_parts.append(assistants_part) + # Add cache point block for assistant content elements + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + assistants_parts.append(_cache_point_block) + assistant_content.extend(assistants_parts) + elif _assistant_content is not None and isinstance(_assistant_content, str): + # Skip completely empty strings to avoid blank content blocks + if _assistant_content.strip(): + assistant_content.append(BedrockContentBlock(text=_assistant_content)) + # If content is empty/whitespace, skip it (don't add a placeholder) + # Add cache point block for assistant string content + assistant_message_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + if assistant_message_cache_point_block is not None: + assistant_content.append(assistant_message_cache_point_block) + + _tool_calls = assistant_message_block.get("tool_calls", []) + if _tool_calls: + assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + assistant_list_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + if assistant_list_cache_point_block is not None: + assistant_content.append(assistant_list_cache_point_block) + + msg_i += 1 + + assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") + assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) + + if assistant_content: + contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) + + @staticmethod + def translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks: list[ChatCompletionThinkingBlock], + ) -> list[BedrockContentBlock]: + reasoning_content_blocks: Final[list[BedrockContentBlock]] = [] + for thinking_block in thinking_blocks: + reasoning_text = thinking_block.get("thinking") + reasoning_signature = thinking_block.get("signature") + text_block = BedrockConverseReasoningTextBlock( + text=reasoning_text or "", + ) + if reasoning_signature is not None: + text_block["signature"] = reasoning_signature + reasoning_content_block = BedrockConverseReasoningContentBlock( + reasoningText=text_block, + ) + bedrock_content_block = BedrockContentBlock(reasoningContent=reasoning_content_block) + reasoning_content_blocks.append(bedrock_content_block) + return reasoning_content_blocks + + @staticmethod + def _process_file_message(message: ChatCompletionFileObject) -> BedrockContentBlock: + file_message: Final = message.get("file") + if file_message is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="bedrock", + ) + file_data: Final = file_message.get("file_data") + file_id: Final = file_message.get("file_id") + + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={message}", + model="", + llm_provider="bedrock", + ) + format: Final = file_message.get("format") + return BedrockImageProcessor.process_image_sync(image_url=cast(str, file_id or file_data), format=format) + + @staticmethod + async def _async_process_file_message( + message: ChatCompletionFileObject, + ) -> BedrockContentBlock: + file_message: Final = message.get("file") + if file_message is None: + raise litellm.BadRequestError( + message="Content block has type='file' but is missing the required 'file' field", + model=None, + llm_provider="bedrock", + ) + file_data: Final = file_message.get("file_data") + file_id: Final = file_message.get("file_id") + format: Final = file_message.get("format") + if file_data is None and file_id is None: + raise litellm.BadRequestError( + message=f"file_data and file_id cannot both be None. Got={message}", + model="", + llm_provider="bedrock", + ) + return await BedrockImageProcessor.process_image_async(image_url=cast(str, file_id or file_data), format=format) + + @staticmethod + def _process_document_message(element: dict) -> BedrockContentBlock: + """Convert a document content block to a Bedrock DocumentBlock. + + Handles the Anthropic-style document format: + {"type": "document", "source": {"type": "base64", "media_type": "application/pdf", "data": "..."}} + """ + source: Final = element["source"] + source_type: Final = source.get("type") + if source_type != "base64": + raise ValueError( + f"Bedrock Converse only supports base64-encoded document sources, got '{source_type}'. " + "Please convert the document to base64 before sending to Bedrock." + ) + media_type: Final[str] = source["media_type"] + data: Final[str] = source["data"] + doc_format = BedrockImageProcessor._validate_format(mime_type=media_type, image_format=media_type.split("/")[1]) + + # Deterministic name using the same hashing pattern as _create_bedrock_block + HASH_SAMPLE_BYTES: Final = 64 * 1024 + normalized: Final = "".join(data.split()).encode("utf-8") + sample: Final = normalized[:HASH_SAMPLE_BYTES] + hasher: Final = hashlib.sha256() + hasher.update(sample) + hasher.update(str(len(normalized)).encode("utf-8")) + content_hash: Final = hasher.hexdigest()[:16] + document_name: Final = f"Document_{content_hash}_{doc_format}" + + return BedrockContentBlock( + document=BedrockDocumentBlock( + source=BedrockSourceBlock(bytes=data), + format=doc_format, + name=document_name, + ) + ) + + @staticmethod + def add_thinking_blocks_to_assistant_content( + thinking_blocks: list[BedrockContentBlock], + assistant_parts: list[BedrockContentBlock], + ) -> list[BedrockContentBlock]: + """ + If contains 'signature', it is a thinking block. + If missing 'signature', it is a text block - e.g. when using a non-anthropic model. + + Handle error raised by bedrock if thinking blocks are provided for a non-thinking model (e.g. nova with tool use) + + Relevant Issue: https://github.com/BerriAI/litellm/issues/9063 + """ + filtered_thinking_blocks: Final = [] + for block in thinking_blocks: + reasoning_content = block.get("reasoningContent", None) + reasoning_text = reasoning_content.get("reasoningText", None) if reasoning_content is not None else None + if reasoning_text and not reasoning_text.get("signature"): + reasoning_text_text = reasoning_text["text"] + if reasoning_text_text.strip(): + assistants_part = BedrockContentBlock(text=reasoning_text_text) + assistant_parts.append(assistants_part) + else: + filtered_thinking_blocks.append(block) + if len(filtered_thinking_blocks) > 0: + assistant_parts.extend(filtered_thinking_blocks) + return assistant_parts + + +def _bedrock_converse_messages_pt( + messages: list, + model: str, + llm_provider: str, + user_continue_message: ChatCompletionUserMessage | None = None, + assistant_continue_message: str | ChatCompletionAssistantMessage | None = None, +) -> list[BedrockMessageBlock]: + """ + Converts given messages from OpenAI format to Bedrock format + + - Roles must alternate b/w 'user' and 'model' (same as anthropic -> merge consecutive roles) + - Please ensure that function response turn comes immediately after a function call turn + - Conversation blocks and tool result blocks cannot be provided in the same turn. Issue: https://github.com/BerriAI/litellm/issues/6053 + """ + + contents: list[BedrockMessageBlock] = [] + msg_i = 0 + + messages = BedrockConverseMessagesProcessor._initial_message_setup( + messages, model, llm_provider, user_continue_message + ) + + while msg_i < len(messages): + user_content: list[BedrockContentBlock] = [] + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "user": + message_block = get_user_message_block_or_continue_message( + message=messages[msg_i], + user_continue_message=user_continue_message, + ) + if isinstance(message_block["content"], list): + _parts: list[BedrockContentBlock] = [] + for element in message_block["content"]: + if isinstance(element, dict): + if element["type"] == "text": + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "guarded_text": + # Wrap guarded_text in guardContent block + _part = BedrockContentBlock(guardContent={"text": {"text": element["text"]}}) + _parts.append(_part) + elif element["type"] in ("grounding_source", "query"): + # Contextual grounding tags are guardrail metadata; the + # model only needs the underlying text, so render them as + # plain text on the generate path. + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "image_url": + format: str | None = None + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + format = element["image_url"].get("format") + else: + image_url = element["image_url"] + _part = BedrockImageProcessor.process_image_sync( + image_url=image_url, + format=format, + ) + _parts.append(_part) + elif element["type"] == "file": + _part = BedrockConverseMessagesProcessor._process_file_message( + message=cast(ChatCompletionFileObject, element) + ) + _parts.append(_part) + elif element["type"] == "document": + _part = BedrockConverseMessagesProcessor._process_document_message(element) + _parts.append(_part) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + if ( + _parts + and message_block.get("cache_control") is not None + and not any("cachePoint" in part for part in _parts) + ): + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + user_content.extend(_parts) + elif message_block["content"] and isinstance(message_block["content"], str): + _part = BedrockContentBlock(text=messages[msg_i]["content"]) + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block, block_type="content_block", model=model + ) + user_content.append(_part) + if _cache_point_block is not None: + user_content.append(_cache_point_block) + + msg_i += 1 + if user_content: + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=user_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(user_content) + else: + contents.append(BedrockMessageBlock(role="user", content=user_content)) + + ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## + tool_content: list[BedrockContentBlock] = [] + while msg_i < len(messages) and messages[msg_i]["role"] == "tool": + tool_call_result = _convert_to_bedrock_tool_call_result(messages[msg_i]) + current_message = messages[msg_i] + + # Add the tool result first + tool_content.append(tool_call_result) + + # Check if we need to add a separate cachePoint block + tool_msg_cache_control = None + + # Check for message-level cache_control + if current_message.get("cache_control", None) is not None: + tool_msg_cache_control = current_message["cache_control"] + # Check for content-level cache_control in list content + elif isinstance(current_message.get("content"), list): + for content_element in current_message["content"]: + if isinstance(content_element, dict) and content_element.get("cache_control", None) is not None: + tool_msg_cache_control = content_element["cache_control"] + break + + # Add a separate cachePoint block if cache_control is present + if tool_msg_cache_control is not None: + cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + {"cache_control": tool_msg_cache_control}, + block_type="content_block", + model=model, + ) + if cache_point_block is not None: + tool_content.append(cache_point_block) + + msg_i += 1 + # Deduplicate toolResult blocks with the same toolUseId + tool_content = _deduplicate_bedrock_tool_content(tool_content) + if tool_content: + # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) + if len(contents) > 0 and contents[-1]["role"] == "user": + if assistant_continue_message is not None or litellm.modify_params is True: + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(tool_content) + else: + contents.append(BedrockMessageBlock(role="user", content=tool_content)) + assistant_content: list[BedrockContentBlock] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_message_block = get_assistant_message_block_or_continue_message( + message=messages[msg_i], + assistant_continue_message=assistant_continue_message, + ) + assistant_message_content_start: int = len(assistant_content) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + _assistant_content = assistant_message_block.get("content", None) + thinking_blocks = cast( + list[ChatCompletionThinkingBlock] | None, + assistant_message_block.get("thinking_blocks"), + ) + + if thinking_blocks is not None: + converted_thinking_blocks = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, + ) + + if _assistant_content is not None and isinstance(_assistant_content, list): + assistants_parts: list[BedrockContentBlock] = [] + for element in _assistant_content: + if isinstance(element, dict): + if element["type"] == "thinking": + thinking_block = ( + BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks=[cast(ChatCompletionThinkingBlock, element)] + ) + ) + assistants_parts = ( + BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) + ) + elif element["type"] == "text": + # AWS Bedrock doesn't allow empty or whitespace-only text content + # Skip completely empty strings to avoid blank content blocks + if element.get("text", "").strip(): + assistants_part = BedrockContentBlock(text=element["text"]) + assistants_parts.append(assistants_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + assistants_part = BedrockImageProcessor.process_image_sync(image_url=image_url) + assistants_parts.append(assistants_part) + # Add cache point block for assistant content elements + _cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + message_block=cast(OpenAIMessageContentListBlock, element), + block_type="content_block", + model=model, + ) + if _cache_point_block is not None: + assistants_parts.append(_cache_point_block) + assistant_content.extend(assistants_parts) + elif _assistant_content is not None and isinstance(_assistant_content, str): + # Skip completely empty strings to avoid blank content blocks + if _assistant_content.strip(): + assistant_content.append(BedrockContentBlock(text=_assistant_content)) + # Add cache point block for assistant string content + assistant_message_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + if assistant_message_cache_point_block is not None: + assistant_content.append(assistant_message_cache_point_block) + _tool_calls = assistant_message_block.get("tool_calls", []) + if _tool_calls: + assistant_content.extend(_convert_to_bedrock_tool_call_invoke(_tool_calls, model=model)) + + if ( + isinstance(_assistant_content, list) + and len(assistant_content) > assistant_message_content_start + and assistant_message_block.get("cache_control") is not None + and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) + ): + assistant_list_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message + if assistant_list_cache_point_block is not None: + assistant_content.append(assistant_list_cache_point_block) + + msg_i += 1 + + assistant_content = _deduplicate_bedrock_content_blocks(assistant_content, "toolUse") + assistant_content = _sort_bedrock_assistant_content_blocks(assistant_content) + + if assistant_content: + contents.append(BedrockMessageBlock(role="assistant", content=assistant_content)) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return _ensure_document_messages_have_text(_rename_duplicate_bedrock_document_names(contents)) + + +def make_valid_bedrock_tool_name(input_tool_name: str) -> str: + """Normalize tool names to Bedrock pattern [a-zA-Z][a-zA-Z0-9_-]*.""" + + def replace_invalid(char): + if char.isalnum() or char in ("_", "-"): + return char + return "_" + + # If the string is empty, return a default valid identifier + if input_tool_name is None or len(input_tool_name) == 0: + return input_tool_name + bedrock_tool_name = copy.copy(input_tool_name) + # If it doesn't start with a letter, prepend 'a' + if not bedrock_tool_name[0].isalpha(): + bedrock_tool_name = "a" + bedrock_tool_name + + # Replace any invalid characters with underscores + valid_string: Final = "".join(replace_invalid(char) for char in bedrock_tool_name) + + if input_tool_name != valid_string: + # passed tool name was formatted to become valid + # store it internally so we can use for the response + litellm.bedrock_tool_name_mappings.set_cache(key=valid_string, value=input_tool_name) + + return valid_string + + +def add_cache_point_tool_block(tool: dict, model: str | None = None) -> BedrockToolBlock | None: + from litellm.llms.bedrock.common_utils import ( + bedrock_model_accepts_cache_points, + is_claude_4_5_on_bedrock, + ) + + cache_control: Final = tool.get("cache_control", None) + if cache_control is not None and bedrock_model_accepts_cache_points(model): + cache_point: Final = cache_control.get("type", "ephemeral") + if cache_point == "ephemeral": + cache_point_block: Final[CachePointBlock] = {"type": "default"} + if isinstance(cache_control, dict) and "ttl" in cache_control: + ttl: Final = cache_control["ttl"] + if ttl in ["5m", "1h"] and model is not None and is_claude_4_5_on_bedrock(model): + cache_point_block["ttl"] = ttl + return {"cachePoint": cache_point_block} + return None + + +def _is_bedrock_tool_block(tool: dict) -> bool: + """ + Check if a tool is already a BedrockToolBlock. + + BedrockToolBlock has one of: systemTool, toolSpec, or cachePoint. + This is used to detect tools that are already in Bedrock format + (e.g., systemTool for Nova grounding) vs OpenAI-style function tools + that need transformation. + + Args: + tool: The tool dict to check + + Returns: + True if the tool is already a BedrockToolBlock, False otherwise + + Examples: + >>> _is_bedrock_tool_block({"systemTool": {"name": "nova_grounding"}}) + True + >>> _is_bedrock_tool_block({"type": "function", "function": {...}}) + False + """ + return isinstance(tool, dict) and ("systemTool" in tool or "toolSpec" in tool or "cachePoint" in tool) + + +def _bedrock_tools_pt(tools: list, model: str | None = None) -> list[BedrockToolBlock]: + """ + OpenAI tools looks like: + tools = [ + { + "type": "function", + "function": { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + } + } + ] + """ + """ + Bedrock toolConfig looks like: + "tools": [ + { + "toolSpec": { + "name": "top_song", + "description": "Get the most popular song played on a radio station.", + "inputSchema": { + "json": { + "type": "object", + "properties": { + "sign": { + "type": "string", + "description": "The call sign for the radio station for which you want the most popular song. Example calls signs are WZPZ, and WKRP." + } + }, + "required": [ + "sign" + ] + } + } + } + } + ] + """ + from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs + from litellm.llms.bedrock.common_utils import ( + bedrock_converse_supports_strict_tools, + normalize_json_schema_custom_types_to_object, + ) + + _valid_json_schema_root_types = frozenset(("array", "boolean", "integer", "null", "number", "object", "string")) + # Only Claude on Bedrock honours strict tool schemas; other families + # (Nova, Llama, GPT-OSS) reject the strict field outright. Opus 4.7/4.8 + # also reject `strict` on Bedrock Converse (see #31582) — their validator + # maps toolSpec to the native Anthropic tool shape, which has no strict + # field, even though Anthropic's native API accepts it as a top-level key. + supports_strict_tools: Final = bool(model and bedrock_converse_supports_strict_tools(model)) + tool_block_list: Final[list[BedrockToolBlock]] = [] + for tool_idx, tool in enumerate(tools): + # Check if tool is already a BedrockToolBlock (e.g., systemTool for Nova grounding) + if _is_bedrock_tool_block(tool): + # Already a BedrockToolBlock, pass it through + tool_block_list.append(tool) + continue + + # Responses built-in tools (web_search, image_generation, namespace, tool_search, + # custom) carry neither an OpenAI "function" nor an Anthropic "input_schema" and have + # no Bedrock toolSpec equivalent; drop them instead of emitting an empty junk toolSpec. + if isinstance(tool, dict) and "function" not in tool and "input_schema" not in tool: + continue + + # OpenAI function tools, or Anthropic Messages / Claude Code ({name, input_schema, type, ...}) + if isinstance(tool, dict) and "input_schema" in tool and "function" not in tool: + parameters = copy.deepcopy(tool.get("input_schema") or {"type": "object", "properties": {}}) + raw_name = tool.get("name", "") or "" + _tool_description = tool.get("description", None) + else: + parameters = copy.deepcopy(tool.get("function", {}).get("parameters", {"type": "object", "properties": {}})) + raw_name = tool.get("function", {}).get("name", "") or "" + _tool_description = tool.get("function", {}).get("description", None) + + if not (raw_name and str(raw_name).strip()): + raw_name = f"litellm_unnamed_tool_{tool_idx}" + + # related issue: https://github.com/BerriAI/litellm/issues/5007 + # Bedrock tool names must satisfy pattern: [a-zA-Z][a-zA-Z0-9_-]* + name = make_valid_bedrock_tool_name(input_tool_name=raw_name) + if _tool_description: # bedrock doesn't accept empty "" or None descriptions + description = _tool_description + else: + description = name + + defs = parameters.pop("$defs", {}) + defs_copy = copy.deepcopy(defs) + # Expand $ref references in parameters using the definitions + # Note: We don't pre-flatten defs as that causes exponential memory growth + # with circular references (see issue #19098). unpack_defs handles nested + # refs recursively and correctly detects/skips circular references. + unpack_defs(parameters, defs_copy) + normalize_json_schema_custom_types_to_object(parameters) + if parameters.get("type") not in _valid_json_schema_root_types: + parameters["type"] = "object" + tool_block = cast( + BedrockToolBlock, + BedrockToolSpec( + name=name, + description=description, + parameters=parameters, + strict=tool.get("function", {}).get("strict", None), + supports_strict_tools=supports_strict_tools, + ), + ) + tool_block_list.append(tool_block) + + ## ADD CACHE POINT TOOL BLOCK ## + cache_point_tool_block = add_cache_point_tool_block(tool, model=model) + if cache_point_tool_block is not None: + tool_block_list.append(cache_point_tool_block) + + return tool_block_list + + +# Function call template +def function_call_prompt(messages: list, functions: list): + function_prompt = """Produce JSON OUTPUT ONLY! Adhere to this format {"name": "function_name", "arguments":{"argument_name": "argument_value"}} The following functions are available to you:""" + for function in functions: + function_prompt += f"""\n{function}\n""" + + function_added_to_prompt = False + for message in messages: + if "system" in message["role"]: + if isinstance(message["content"], str): + message["content"] += f""" {function_prompt}""" + else: + message["content"].append({"type": "text", "text": f""" {function_prompt}"""}) + function_added_to_prompt = True + + if function_added_to_prompt is False: + messages.append({"role": "system", "content": f"""{function_prompt}"""}) + + return messages + + +def response_schema_prompt(model: str, response_schema: dict) -> str: + """ + Decides if a user-defined custom prompt or default needs to be used + + Returns the prompt str that's passed to the model as a user message + """ + custom_prompt_details: dict | None = None + response_schema_as_message: Final = [{"role": "user", "content": f"{response_schema}"}] + if f"{model}/response_schema_prompt" in litellm.custom_prompt_dict: + custom_prompt_details = litellm.custom_prompt_dict[ + f"{model}/response_schema_prompt" + ] # allow user to define custom response schema prompt by model + elif "response_schema_prompt" in litellm.custom_prompt_dict: + custom_prompt_details = litellm.custom_prompt_dict["response_schema_prompt"] + + if custom_prompt_details is not None: + return custom_prompt( + role_dict=custom_prompt_details["roles"], + initial_prompt_value=custom_prompt_details["initial_prompt_value"], + final_prompt_value=custom_prompt_details["final_prompt_value"], + messages=response_schema_as_message, + ) + else: + return default_response_schema_prompt(response_schema=response_schema) + + +def default_response_schema_prompt(response_schema: dict) -> str: + """ + Used if provider/model doesn't support 'response_schema' param. + + This is the default prompt. Allow user to override this with a custom_prompt. + """ + prompt_str: Final = f"""Use this JSON schema: + ```json + {response_schema} + ```""" + return prompt_str + + +# Custom prompt template +def custom_prompt( + role_dict: dict, + messages: list, + initial_prompt_value: str = "", + final_prompt_value: str = "", + bos_token: str = "", + eos_token: str = "", +) -> str: + prompt = bos_token + initial_prompt_value + bos_open = True + ## a bos token is at the start of a system / human message + ## an eos token is at the end of the assistant response to the message + for message in messages: + role = message["role"] + + if role in ["system", "human"] and not bos_open: + prompt += bos_token + bos_open = True + + pre_message_str = ( + role_dict[role]["pre_message"] if role in role_dict and "pre_message" in role_dict[role] else "" + ) + post_message_str = ( + role_dict[role]["post_message"] if role in role_dict and "post_message" in role_dict[role] else "" + ) + if isinstance(message["content"], str): + prompt += pre_message_str + message["content"] + post_message_str + elif isinstance(message["content"], list): + text_str = "" + for content in message["content"]: + if content.get("text", None) is not None and isinstance(content["text"], str): + text_str += content["text"] + prompt += pre_message_str + text_str + post_message_str + + if role == "assistant": + prompt += eos_token + bos_open = False + + prompt += final_prompt_value + return prompt + + +def prompt_factory( + model: str, + messages: list, + custom_llm_provider: str | None = None, + api_key: str | None = None, +): + original_model_name: Final = model + model = model.lower() + if custom_llm_provider == "ollama": + return ollama_pt(model=model, messages=messages) + elif custom_llm_provider == "anthropic": + if litellm.AnthropicTextConfig._is_anthropic_text_model(model): + return anthropic_pt(messages=messages) + return anthropic_messages_pt(messages=messages, model=model, llm_provider=custom_llm_provider) + elif custom_llm_provider == "anthropic_xml": + return anthropic_messages_pt_xml(messages=messages) + elif custom_llm_provider == "gemini": + if ( + model == "gemini-pro-vision" + or litellm.supports_vision(model=model) + or litellm.supports_vision(model=custom_llm_provider + "/" + model) + ): + return _gemini_vision_convert_messages(messages=messages) + else: + return gemini_text_image_pt(messages=messages) + elif custom_llm_provider == "mistral": + return litellm.MistralConfig()._transform_messages(messages=messages, model=model) + elif custom_llm_provider == "bedrock": + if "amazon.titan-text" in model: + return amazon_titan_pt(messages=messages) + elif "anthropic." in model: + if any(_ in model for _ in ["claude-2.1", "claude-v2:1"]): + return claude_2_1_pt(messages=messages) + else: + return anthropic_pt(messages=messages) + elif "mistral." in model: + return mistral_instruct_pt(messages=messages) + elif "llama2" in model and "chat" in model: + return llama_2_chat_pt(messages=messages) + elif ("llama3" in model or "llama4" in model) and "instruct" in model: + return hf_chat_template( + model="meta-llama/Meta-Llama-3-8B-Instruct", + messages=messages, + ) + + elif custom_llm_provider == "clarifai": + if "claude" in model: + return anthropic_pt(messages=messages) + + elif custom_llm_provider == "perplexity": + for message in messages: + message.pop("name", None) + return messages + elif custom_llm_provider == "azure_text": + return azure_text_pt(messages=messages) + elif custom_llm_provider == "watsonx": + from litellm.llms.watsonx.chat.transformation import IBMWatsonXChatConfig + + return IBMWatsonXChatConfig.apply_prompt_template(model=model, messages=messages) + + try: + if "meta-llama/llama-2" in model and "chat" in model: + return llama_2_chat_pt(messages=messages) + elif ("meta-llama/llama-3" in model or "meta-llama-3" in model) and "instruct" in model: + return hf_chat_template( + model="meta-llama/Meta-Llama-3-8B-Instruct", + messages=messages, + ) + elif ( + "tiiuae/falcon" in model + ): # Note: for the instruct models, it's best to use a User: .., Assistant:.. approach in your prompt template. + if model == "tiiuae/falcon-180B-chat": + return falcon_chat_pt(messages=messages) + elif "instruct" in model: + return falcon_instruct_pt(messages=messages) + elif "mosaicml/mpt" in model: + if "chat" in model: + return mpt_chat_pt(messages=messages) + elif "codellama/codellama" in model or "togethercomputer/codellama" in model: + if "instruct" in model: + return llama_2_chat_pt( + messages=messages + ) # https://huggingface.co/blog/codellama#conversational-instructions + elif "wizardlm/wizardcoder" in model: + return wizardcoder_pt(messages=messages) + elif "phind/phind-codellama" in model: + return phind_codellama_pt(messages=messages) + elif "togethercomputer/llama-2" in model and ("instruct" in model or "chat" in model): + return llama_2_chat_pt(messages=messages) + elif model in [ + "gryphe/mythomax-l2-13b", + "gryphe/mythomix-l2-13b", + "gryphe/mythologic-l2-13b", + ]: + return alpaca_pt(messages=messages) + elif model.strip() == "meta-llama/llamaguard-7b": + chat_template = "{% if messages|length % 2 == 0 %}{% set role = 'Agent' %}{% else %}{% set role = 'User' %}{% endif %}{{ \"[INST] Task: Check if there is unsafe content in '\" + role + \"' messages in conversations according our safety policy with the below categories.\n\n\nO1: Violence and Hate. \nShould not\n- Help people plan or engage in violence.\n- Encourage people to engage in violence.\n- Express hateful, derogatory, or demeaning sentiments against people based on sensitive personal characteristics like their race, color, religion, national origin, sexual orientation, gender, gender identity, or disability.\n- Encourage discrimination against people based on those characteristics and should not use slurs or create insults based on those characteristics.\nCan\n- Provide information on violence and discrimination.\n- Discuss the topics of hate and violence, and can discuss historical events involving violence.\nO2: Sexual Content.\nShould not\n- Engage in sexually explicit (i.e., erotic) conversations and should not encourage people to engage in sexual activities.\nCan\n- Discuss topics related to sexuality and sexual education.\n- Provide general information regarding sexual activities.\n- Provide guidance on topics related to sexual health.\nO3: Criminal Planning.\nShould not\n- Help people plan or engage in criminal activities like theft, kidnapping, or financial crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in criminal activities or condone criminal behavior.\nCan\n- Help people understand how crimes are committed, provided that the response is not so detailed that it could be used to provide a roadmap for future crimes.\n- Help people understand how systems work that are used to prevent crimes, provided those descriptions do not provide detailed information on how to defeat those systems.\nO4: Guns and Illegal Weapons.\nShould not\n- Assist in the planning or execution of firearm or illegal weapons crimes, especially when the person indicates that they may act on the information provided to them.\n- Encourage people to engage in firearm or illegal weapons crimes or condone such behavior.\nCan\n- Discuss firearms and the arguments for and against firearm ownership.\nO5: Regulated or Controlled Substances.\nShould not\n- Encourage or assist people in the trafficking, consumption, or use of regulated or controlled substances, including but not limited to illegal drugs, tobacco, alcohol, and cannabis.\n- Assist or encourage people to create such substances when it is illegal to do so.\nCan\n- Discuss the use of regulated or controlled substances, and can provide general information on those substances (ex: what their chemical composition is and what their effects are).\n- Provide information on how those substances are manufactured, provided that the information is not so detailed as to provide step-by-step instructions for their creation.\nO6: Self-Harm.\nShould not\n- Encourage people to harm themselves.\n- Romanticize or condone self-harm or suicide.\n- Provide information regarding the methods of suicide or self-harm.\n- Provide any information or advice, other than information regarding appropriate health resources, to people who express or imply that they wish to harm themselves.\nCan\n- Provide general information regarding self-harm and suicide (ex: statistics, the mental health challenges faced by those who take these actions).\nShould\n- Provide information regarding appropriate health resources to people who imply or express the desire to harm themselves.\n\n\n\n\n\" }}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{% set role = 'User' %}{% elif message['role'] == 'assistant' %}{% set role = 'Agent' %}{% endif %}{{ role + ': ' + content.strip() + '\n\n' }}{% endfor %}{{ \"\n\nProvide your safety assessment for \" + role + \" in the above conversation:\n- First line must read 'safe' or 'unsafe'.\n- If unsafe, a second line must include a comma-separated list of violated categories. [/INST]\" }}" + return hf_chat_template(model=model, messages=messages, chat_template=chat_template) + else: + return hf_chat_template(original_model_name, messages) + except Exception: + return default_pt( + messages=messages + ) # default that covers Bloom, T-5, any non-chat tuned model (e.g. base Llama2) + + +def get_attribute_or_key(tool_or_function, attribute, default=None): + if hasattr(tool_or_function, attribute): + return getattr(tool_or_function, attribute) + if isinstance(tool_or_function, Mapping): + return tool_or_function.get(attribute, default) + return default + + +class NormalizedToolCall(TypedDict): + id: str | None + name: str | None + arguments: dict[str, object] + + +_ArgumentObjects: TypeAlias = tuple[dict[str, object], ...] +_ParsedToolCall: TypeAlias = tuple[str | None, str | None, _ArgumentObjects] + + +def _optional_call_id(value: object) -> str | None: + if isinstance(value, str) and value: + return value + return None + + +def _optional_tool_name(value: object) -> str | None: + if isinstance(value, str): + return value + return None + + +def _split_tool_call_ids(calls: Sequence[tuple[str | None, int]]) -> tuple[tuple[str | None, ...], ...]: + taken: Final = frozenset(_sanitize_anthropic_tool_use_id(call_id) for call_id, _ in calls if call_id) + + def fresh(call_id: str) -> Iterator[str]: + return filter( + lambda candidate: _sanitize_anthropic_tool_use_id(candidate) not in taken, + (f"{call_id}__concat_{n}" for n in itertools.count(1)), + ) + + suffixes: Final = MappingProxyType( + {_sanitize_anthropic_tool_use_id(call_id): fresh(call_id) for call_id, count in calls if call_id and count > 1} + ) + return tuple( + ( + call_id, + *(next(suffixes[_sanitize_anthropic_tool_use_id(call_id)]) for _ in range(count - 1)), + ) + if call_id + else (None,) * count + for call_id, count in calls + ) + + +def _parse_tool_call_arguments(raw: object, tool_name: str | None, context: str) -> _ArgumentObjects: + # Anthropic's tool_use blocks already carry a parsed dict in "input"; + # chat completions and the Responses API carry a JSON string that may be + # truncated by the model, so route those through the repair-aware parser. + if isinstance(raw, dict): + return (raw,) + if not isinstance(raw, str): + return ({},) + normalized_raw: Final = "{}" if raw == REDACTED_BY_LITELLM else raw + try: + parsed: Final = parse_tool_call_arguments(normalized_raw, tool_name=tool_name, context=context) + except ValueError as e: + salvaged: Final = salvage_concatenated_tool_arguments(normalized_raw) + if salvaged: + verbose_logger.warning( + "Recovered %d tool call(s) from concatenated JSON arguments for tool '%s' (%s)", + len(salvaged), + tool_name or "", + context, + ) + return salvaged + verbose_logger.warning("Failed to parse tool call arguments: %s", e) + return ({},) + return (parsed,) if isinstance(parsed, dict) else ({},) + + +def _choice_tool_calls(choice: object) -> tuple[object, ...]: + message: Final = get_attribute_or_key(choice, "message", None) + tool_calls: Final = get_attribute_or_key(message, "tool_calls", None) if message is not None else None + if isinstance(tool_calls, list): + return tuple(tool_calls) + return () + + +def _selected_choices(response: object, include_all_choices: bool) -> tuple[object, ...]: + choices: Final = get_attribute_or_key(response, "choices", None) + if not isinstance(choices, list) or not choices: + return () + if include_all_choices: + return tuple(choices) + return (choices[0],) + + +def _parsed_chat_tool_call(tool_call: object) -> _ParsedToolCall | None: + function: Final = get_attribute_or_key(tool_call, "function", None) + if function is None: + return None + name: Final = _optional_tool_name(get_attribute_or_key(function, "name")) + return ( + _optional_call_id(get_attribute_or_key(tool_call, "id")), + name, + _parse_tool_call_arguments( + get_attribute_or_key(function, "arguments", "{}"), + tool_name=name, + context="chat completions", + ), + ) + + +def _parsed_calls_in_choice(choice: object) -> tuple[_ParsedToolCall, ...]: + return tuple( + parsed for tool_call in _choice_tool_calls(choice) if (parsed := _parsed_chat_tool_call(tool_call)) is not None + ) + + +def _parsed_chat_tool_calls(response: object, include_all_choices: bool) -> tuple[_ParsedToolCall, ...]: + grouped: Final = tuple( + _parsed_calls_in_choice(choice) for choice in _selected_choices(response, include_all_choices) + ) + return tuple(itertools.chain.from_iterable(grouped)) + + +def _normalized_tool_calls_for_parse( + name: str | None, + call_ids: tuple[str | None, ...], + arguments: _ArgumentObjects, +) -> tuple[NormalizedToolCall, ...]: + return tuple( + NormalizedToolCall(id=call_id, name=name, arguments=argument) + for call_id, argument in zip(call_ids, arguments, strict=True) + ) + + +def _normalized_tool_calls_from_parses(parses: Sequence[_ParsedToolCall]) -> tuple[NormalizedToolCall, ...]: + id_groups: Final = _split_tool_call_ids(tuple((call_id, len(arguments)) for call_id, _, arguments in parses)) + grouped: Final = tuple( + _normalized_tool_calls_for_parse(name, call_ids, arguments) + for (_, name, arguments), call_ids in zip(parses, id_groups, strict=True) + ) + return tuple(itertools.chain.from_iterable(grouped)) + + +def _tool_calls_from_chat_completion_response( + response: object, include_all_choices: bool = False +) -> tuple[NormalizedToolCall, ...]: + return _normalized_tool_calls_from_parses(_parsed_chat_tool_calls(response, include_all_choices)) + + +def _response_function_calls(response: object) -> tuple[object, ...]: + output: Final = get_attribute_or_key(response, "output", None) + if not isinstance(output, list): + return () + return tuple(item for item in output if get_attribute_or_key(item, "type") == "function_call") + + +def _parsed_response_tool_call(item: object) -> _ParsedToolCall: + name: Final = _optional_tool_name(get_attribute_or_key(item, "name")) + raw_id: Final = get_attribute_or_key(item, "call_id") or get_attribute_or_key(item, "id") + return ( + _optional_call_id(raw_id), + name, + _parse_tool_call_arguments( + get_attribute_or_key(item, "arguments", "{}"), + tool_name=name, + context="responses API", + ), + ) + + +def _tool_calls_from_responses_api_response(response: object) -> tuple[NormalizedToolCall, ...]: + parses: Final = tuple(_parsed_response_tool_call(item) for item in _response_function_calls(response)) + return _normalized_tool_calls_from_parses(parses) + + +def _tool_calls_from_anthropic_messages_response(response: object) -> list[NormalizedToolCall]: + content: Final = get_attribute_or_key(response, "content", None) + if not isinstance(content, list): + return [] + result: Final[list[NormalizedToolCall]] = [] + for block in content: + if get_attribute_or_key(block, "type") != "tool_use": + continue + raw_input = get_attribute_or_key(block, "input", {}) + result.append( + NormalizedToolCall( + id=get_attribute_or_key(block, "id"), + name=get_attribute_or_key(block, "name"), + arguments=raw_input if isinstance(raw_input, dict) else {}, + ) + ) + return result + + +def get_tool_calls_from_response(response: object, include_all_choices: bool = False) -> list[NormalizedToolCall]: + """ + Extract tool/function calls from a response object into a normalized + ``{"id", "name", "arguments"}`` shape, regardless of which API surface + produced it: chat completions (``choices[].message.tool_calls``), + the Responses API (``output`` items of type ``function_call``), or the + Anthropic Messages API (``content`` blocks of type ``tool_use``). + + ``include_all_choices`` decides the chat-completions scope: the default + reads only ``choices[0]``, which is what consumers that act on THE reply + (e.g. guardrails rebuilding the primary assistant message) want; usage + accounting passes True because every choice of an ``n>1`` request costs + money and its tool calls really ran. The other surfaces have a single + output, so the flag has no effect on them. + + Callers that only care about a specific tool should filter the result by + ``name`` themselves -- this returns every tool call found. + """ + chat_tool_calls: Final = _tool_calls_from_chat_completion_response( + response, include_all_choices=include_all_choices + ) + if chat_tool_calls: + return list(chat_tool_calls) + for extractor in ( + _tool_calls_from_responses_api_response, + _tool_calls_from_anthropic_messages_response, + ): + tool_calls = extractor(response) + if tool_calls: + return list(tool_calls) + return [] + + +def has_tool_with_name(tools: object, tool_name: str) -> bool: + """ + Check whether a tools list (as sent to an LLM) includes a tool with the + given name, regardless of shape: OpenAI-style function tools + (``{"type": "function", "function": {"name": ...}}``) or Anthropic's + native tool shape (a top-level ``"name"``, e.g. + ``{"name": ..., "input_schema": ...}``). Anthropic's documented client + tool format doesn't require a ``"type"`` key at all -- ``"custom"`` is + only one of several possible values -- so any non-OpenAI-shaped tool is + matched on its top-level ``"name"``. + """ + if not isinstance(tools, list): + return False + for tool in tools: + if not isinstance(tool, dict): + continue + function = tool.get("function") + if tool.get("type") == "function" and isinstance(function, dict): + if function.get("name") == tool_name: + return True + elif tool.get("name") == tool_name: + return True + return False + + +def resolve_structured_messages( + messages: list[dict[str, object]] | None, + request_kwargs: dict[str, Any], +) -> list[dict[str, object]] | None: + """ + Normalize a request's messages to OpenAI-spec chat-completions shape, + regardless of which API surface produced them (chat completions, + Anthropic /v1/messages, Responses API ``input``, etc). + + Returns ``messages`` unchanged if already present. Otherwise dispatches + through the guardrail translation handlers (the same per-surface + conversion logic guardrails use) to convert e.g. Responses API ``input`` + into a message list. Returns ``None`` if no messages could be resolved. + """ + if messages: + return messages + + from litellm.litellm_core_utils.api_route_to_call_types import ( + get_call_types_for_route, + ) + from litellm.llms import load_guardrail_translation_mappings + from litellm.types.utils import CallTypes + + mappings: Final = load_guardrail_translation_mappings() + call_type: CallTypes | None = None + + # 1. Try route-based inference from proxy metadata + route: Final = request_kwargs.get("litellm_metadata", {}).get("user_api_key_request_route") + if route: + call_types_list: Final = get_call_types_for_route(route) + if call_types_list: + for ct in call_types_list: + if ct in mappings: + call_type = ct + break + + # 2. Fallback: try each mapped handler until one produces messages + handlers_to_try: Final[list[Any]] = [] + if call_type is not None and call_type in mappings: + handlers_to_try.append(mappings[call_type]()) + else: + handlers_to_try.extend(handler_cls() for handler_cls in mappings.values()) + + for handler in handlers_to_try: + structured = handler.get_structured_messages(request_kwargs) + if structured: + return [msg if isinstance(msg, dict) else msg.model_dump() for msg in structured] + return None diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index a83f9ec5a54..08f4833e6a5 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1719,7 +1719,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): anthropic_system_message_content["cache_control"] = system_message_block["cache_control"] anthropic_system_message_list.append(anthropic_system_message_content) elif isinstance(message["content"], list): - content_start: int = len(anthropic_system_message_list) + content_start: int = len(anthropic_system_message_list) # pyright: ignore[reportGeneralTypeIssues] # reset per system message for _content in message["content"]: # Skip empty text blocks - Anthropic API raises errors for empty text text_value = _content.get("text") diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index b310080687b..a32606d31b2 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -185,6 +185,33 @@ async def test_bedrock_assistant_list_preserves_message_cache_control(block_cach assert "toolUse" in blocks[-2] +@pytest.mark.asyncio +async def test_bedrock_assistant_message_cache_point_stays_before_following_assistant_text(): + messages: Final = [ + {"role": "user", "content": "hi"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "cached prefix"}], + "cache_control": {"type": "ephemeral"}, + }, + {"role": "assistant", "content": [{"type": "text", "text": "uncached suffix"}]}, + ] + model: Final = "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + expected: Final = [ + {"text": "cached prefix"}, + {"cachePoint": {"type": "default"}}, + {"text": "uncached suffix"}, + ] + + sync_result: Final = _bedrock_converse_messages_pt(messages=messages, model=model, llm_provider="bedrock") + async_result: Final = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, model=model, llm_provider="bedrock" + ) + + assert sync_result[-1]["content"] == expected + assert async_result[-1]["content"] == expected + + def _get_gemini_function_response_inline_data_parts(result): assert isinstance(result, list), "expected Gemini parts list" assert len(result) == 1, "multimodal function responses should stay in one part" From 11b194e43b3c2ab039f4562ecb13876ebd0c0a9c Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Mon, 28 Sep 2026 00:45:23 +0800 Subject: [PATCH 6/8] fix(anthropic): clarify loop-local cache marker state --- .../prompt_templates/factory.py | 46 ++++++++++--------- 1 file changed, 24 insertions(+), 22 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 9b3cad5db2a..db3b2ce68a1 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -2488,14 +2488,13 @@ def anthropic_messages_pt( for content in user_content[user_content_start:] ) ): - cached_user_content: AnthropicMessagesUserMessageValues = copy.copy(user_content[-1]) # pyright: ignore[reportGeneralTypeIssues] # fresh copy per user message + user_content[-1] = copy.copy(user_content[-1]) add_cache_control_to_content( anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], cached_user_content + dict[str, object], user_content[-1] ), original_content_element=user_message_types_block, ) - user_content[-1] = cached_user_content elif isinstance(user_message_types_block["content"], str): _anthropic_content_text_element: AnthropicMessagesTextParam = { "type": "text", @@ -2799,14 +2798,13 @@ def anthropic_messages_pt( for content in assistant_content[assistant_content_start:] ) ): - cached_assistant_content: AnthropicMessagesAssistantMessageValues = copy.copy(assistant_content[-1]) # pyright: ignore[reportGeneralTypeIssues] # fresh copy per assistant message + assistant_content[-1] = copy.copy(assistant_content[-1]) add_cache_control_to_content( anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], cached_assistant_content + dict[str, object], assistant_content[-1] ), original_content_element=assistant_content_block, ) - assistant_content[-1] = cached_assistant_content msg_i += 1 @@ -4623,10 +4621,11 @@ class BedrockConverseMessagesProcessor: assistant_content.append(BedrockContentBlock(text=_assistant_content)) # If content is empty/whitespace, skip it (don't add a placeholder) # Add cache point block for assistant string content - assistant_message_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message - if assistant_message_cache_point_block is not None: + if ( + assistant_message_cache_point_block := litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + ) is not None: assistant_content.append(assistant_message_cache_point_block) _tool_calls = assistant_message_block.get("tool_calls", []) @@ -4639,10 +4638,11 @@ class BedrockConverseMessagesProcessor: and assistant_message_block.get("cache_control") is not None and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) ): - assistant_list_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message - if assistant_list_cache_point_block is not None: + if ( + assistant_list_cache_point_block := litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + ) is not None: assistant_content.append(assistant_list_cache_point_block) msg_i += 1 @@ -5023,10 +5023,11 @@ def _bedrock_converse_messages_pt( if _assistant_content.strip(): assistant_content.append(BedrockContentBlock(text=_assistant_content)) # Add cache point block for assistant string content - assistant_message_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message - if assistant_message_cache_point_block is not None: + if ( + assistant_message_cache_point_block := litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + ) is not None: assistant_content.append(assistant_message_cache_point_block) _tool_calls = assistant_message_block.get("tool_calls", []) if _tool_calls: @@ -5038,10 +5039,11 @@ def _bedrock_converse_messages_pt( and assistant_message_block.get("cache_control") is not None and not any("cachePoint" in block for block in assistant_content[assistant_message_content_start:]) ): - assistant_list_cache_point_block = litellm.AmazonConverseConfig().get_cache_point_block( - assistant_message_block, block_type="content_block", model=model - ) # pyright: ignore[reportGeneralTypeIssues] # reset per assistant message - if assistant_list_cache_point_block is not None: + if ( + assistant_list_cache_point_block := litellm.AmazonConverseConfig().get_cache_point_block( + assistant_message_block, block_type="content_block", model=model + ) + ) is not None: assistant_content.append(assistant_list_cache_point_block) msg_i += 1 From 8b01e7caaf103ad7bd82157fb22c3e3a7d7b9763 Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Mon, 28 Sep 2026 12:58:46 +0800 Subject: [PATCH 7/8] fix(anthropic): rebuild content lists when applying cache markers --- .../prompt_templates/factory.py | 34 ++++++++++++------- 1 file changed, 22 insertions(+), 12 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index db3b2ce68a1..746622aac29 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -2488,13 +2488,18 @@ def anthropic_messages_pt( for content in user_content[user_content_start:] ) ): - user_content[-1] = copy.copy(user_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], user_content[-1] + user_content = [ # mutable-ok: Anthropic message content requires a list; build a fresh one + *user_content[:-1], + cast( # cast-ok: helper preserves the copied content block type + AnthropicMessagesUserMessageValues, + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], copy.copy(user_content[-1]) + ), + original_content_element=user_message_types_block, + ), ), - original_content_element=user_message_types_block, - ) + ] elif isinstance(user_message_types_block["content"], str): _anthropic_content_text_element: AnthropicMessagesTextParam = { "type": "text", @@ -2798,13 +2803,18 @@ def anthropic_messages_pt( for content in assistant_content[assistant_content_start:] ) ): - assistant_content[-1] = copy.copy(assistant_content[-1]) - add_cache_control_to_content( - anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict - dict[str, object], assistant_content[-1] + assistant_content = [ # mutable-ok: Anthropic message content requires a list; build a fresh one + *assistant_content[:-1], + cast( # cast-ok: helper preserves the copied content block type + AnthropicMessagesAssistantMessageValues, + add_cache_control_to_content( + anthropic_content_element=cast( # cast-ok: copied TypedDict is a dict + dict[str, object], copy.copy(assistant_content[-1]) + ), + original_content_element=assistant_content_block, + ), ), - original_content_element=assistant_content_block, - ) + ] msg_i += 1 From b0898a769e76af08643864d9561f9709f7f1317b Mon Sep 17 00:00:00 2001 From: amazingniuma-art Date: Mon, 28 Sep 2026 15:04:22 +0800 Subject: [PATCH 8/8] test: cover Anthropic and Bedrock cache control paths --- ...llm_core_utils_prompt_templates_factory.py | 83 ++++++++++++++++++- 1 file changed, 81 insertions(+), 2 deletions(-) diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index a32606d31b2..af9b70a017e 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -15,10 +15,11 @@ from litellm.litellm_core_utils.prompt_templates.factory import ( BedrockImageProcessor, _bedrock_converse_messages_pt, _bedrock_tools_pt, - _rename_duplicate_bedrock_document_names, _convert_to_bedrock_tool_call_invoke, - _sanitize_anthropic_tool_use_id, _convert_to_bedrock_tool_call_result, + _rename_duplicate_bedrock_document_names, + _sanitize_anthropic_tool_use_id, + _sort_bedrock_assistant_content_blocks, anthropic_messages_pt, convert_to_anthropic_tool_result, convert_to_gemini_tool_call_result, @@ -55,6 +56,31 @@ def test_anthropic_list_content_preserves_message_cache_control(role, block_cach assert result[-1]["content"] == expected +def test_anthropic_list_message_cache_control_skips_separate_thinking_blocks(): + thinking_block: Final = { + "type": "thinking", + "thinking": "reasoning", + "signature": "valid-signature", + } + cache_control: Final = {"type": "ephemeral"} + messages: Final = [ + {"role": "user", "content": "hi"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "answer"}], + "thinking_blocks": [thinking_block], + "cache_control": cache_control, + }, + ] + + result: Final = anthropic_messages_pt(messages=messages, model="claude-sonnet-4-6", llm_provider="anthropic") + + assert result[-1]["content"] == [ + thinking_block, + {"type": "text", "text": "answer", "cache_control": cache_control}, + ] + + @pytest.mark.parametrize("call_field", ["tool_calls", "function_call"]) def test_anthropic_list_message_cache_control_follows_tool_calls(call_field): cache_control: Final = {"type": "ephemeral"} @@ -185,6 +211,24 @@ async def test_bedrock_assistant_list_preserves_message_cache_control(block_cach assert "toolUse" in blocks[-2] +@pytest.mark.asyncio +async def test_bedrock_assistant_string_preserves_message_cache_control(): + messages: Final = [ + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "checking", "cache_control": {"type": "ephemeral"}}, + ] + model: Final = "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + expected: Final = [{"text": "checking"}, {"cachePoint": {"type": "default"}}] + + sync_result: Final = _bedrock_converse_messages_pt(messages=messages, model=model, llm_provider="bedrock") + async_result: Final = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, model=model, llm_provider="bedrock" + ) + + assert sync_result[-1]["content"] == expected + assert async_result[-1]["content"] == expected + + @pytest.mark.asyncio async def test_bedrock_assistant_message_cache_point_stays_before_following_assistant_text(): messages: Final = [ @@ -212,6 +256,41 @@ async def test_bedrock_assistant_message_cache_point_stays_before_following_assi assert async_result[-1]["content"] == expected +def test_bedrock_assistant_content_sorting_keeps_cache_point_boundaries(): + first_cache_point: Final = {"cachePoint": {"type": "default"}} + second_cache_point: Final = {"cachePoint": {"type": "default", "ttl": "5m"}} + before_cache_point: Final = {"text": "before cache point"} + after_first_cache_point: Final = {"text": "after first cache point"} + after_second_cache_point: Final = {"text": "after second cache point"} + first_tool_use: Final = {"toolUse": {"toolUseId": "call_1", "name": "first"}} + second_tool_use: Final = {"toolUse": {"toolUseId": "call_2", "name": "second"}} + reasoning: Final = {"reasoningContent": {"reasoningText": {"text": "reasoning"}}} + + result: Final = _sort_bedrock_assistant_content_blocks( + [ + first_tool_use, + before_cache_point, + first_cache_point, + second_tool_use, + after_first_cache_point, + reasoning, + second_cache_point, + after_second_cache_point, + ] + ) + + assert result == [ + before_cache_point, + first_tool_use, + first_cache_point, + reasoning, + after_first_cache_point, + second_tool_use, + second_cache_point, + after_second_cache_point, + ] + + def _get_gemini_function_response_inline_data_parts(result): assert isinstance(result, list), "expected Gemini parts list" assert len(result) == 1, "multimodal function responses should stay in one part"