From 5003a3a11be2119edda57eaaf8bfbeb370901559 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 17:52:17 -0700 Subject: [PATCH] fix(bedrock): keep tool_reference results as text on converse (#44590) * test(e2e): cover tool_reference tool results on bedrock converse Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): keep tool_reference results as text on converse Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(bedrock): format tool reference parser Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): cover bedrock converse tool_result of only tool_reference blocks Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): tag converse tool_reference case with e2e metadata Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): avoid unchecked cast in converse tool_reference mapping Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(anthropic): keep tool_reference results as text on responses paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(anthropic): tighten tool_reference mapping on responses paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(completion-extras): expect tool_reference names in responses input Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * revert(bedrock): drop responses converter changes, keep converse tool_reference fix only Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: kerry --- .../prompt_templates/factory.py | 5 + .../test_messages_bedrock_e2e.py | 78 ++++++++- tests/integration/proxy_config.yaml | 6 + ...t_messages_tool_search_bedrock_converse.py | 156 ++++++++++++++++++ ...llm_core_utils_prompt_templates_factory.py | 35 ++++ 5 files changed, 279 insertions(+), 1 deletion(-) create mode 100644 tests/integration/translation/messages/tool_search/test_messages_tool_search_bedrock_converse.py diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 3b0cf0ce238..5d6c1ac0e4b 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -3897,6 +3897,11 @@ def _parse_bedrock_tool_result_content_list( _append_bedrock_tool_result_image_url_block(tool_result_content_blocks, content) elif content["type"] == "file": _append_bedrock_tool_result_file_block(tool_result_content_blocks, content) + elif content["type"] == "tool_reference": + content_map = cast(dict[str, object], content) # cast-ok: untyped anthropic json + tool_name = content_map.get("tool_name") + if isinstance(tool_name, str): + tool_result_content_blocks.append(BedrockToolResultContentBlock(text=tool_name)) return tool_result_content_blocks diff --git a/tests/e2e/llm_translation/test_messages_bedrock_e2e.py b/tests/e2e/llm_translation/test_messages_bedrock_e2e.py index 4e37f01886a..31daa9bba32 100644 --- a/tests/e2e/llm_translation/test_messages_bedrock_e2e.py +++ b/tests/e2e/llm_translation/test_messages_bedrock_e2e.py @@ -3,7 +3,7 @@ from __future__ import annotations from typing import Final import pytest -from anthropic.types import RawContentBlockDeltaEvent, RawMessageDeltaEvent, TextBlock, TextDelta +from anthropic.types import RawContentBlockDeltaEvent, RawMessageDeltaEvent, TextBlock, TextDelta, ToolParam from e2e_config import unique_marker from e2e_metadata import Capability, Domain, Mode, Provider, Route, Subject, meta from lifecycle import ResourceManager @@ -14,9 +14,38 @@ from structured_output import SENTIMENT_OUTPUT_FORMAT, SENTIMENT_PROMPT, assert_ pytestmark = pytest.mark.e2e +GPT_6_1_SOL_BACKEND: Final = "bedrock/global.openai.gpt-6.1-sol" CONVERSE_CLAUDE_BACKEND: Final = "bedrock/converse/us.anthropic.claude-haiku-4-5-20251001-v1:0" NOVA_BACKEND: Final = "bedrock/us.amazon.nova-2-lite-v1:0" +TOOL_SEARCH: Final[ToolParam] = { + "name": "ToolSearch", + "description": "Find available tools by query.", + "input_schema": { + "type": "object", + "properties": {"q": {"type": "string"}}, + "required": ["q"], + }, +} +WEB_FETCH: Final[ToolParam] = { + "name": "WebFetch", + "description": "Fetch a web page.", + "input_schema": { + "type": "object", + "properties": {"url": {"type": "string"}, "prompt": {"type": "string"}}, + "required": ["url", "prompt"], + }, +} +WEB_SEARCH: Final[ToolParam] = { + "name": "WebSearch", + "description": "Search the web.", + "input_schema": { + "type": "object", + "properties": {"query": {"type": "string"}}, + "required": ["query"], + }, +} + def _register(proxy: ProxyClient, resources: ResourceManager, backend: str) -> str: model = f"e2e-messages-bedrock-{unique_marker()}" @@ -34,6 +63,53 @@ def _register(proxy: ProxyClient, resources: ResourceManager, backend: str) -> s class TestBedrockMessages: + @meta( + Subject( + domain=Domain.LLM_TRANSLATION, + route=Route.MESSAGES, + providers=(Provider.BEDROCK,), + models=(GPT_6_1_SOL_BACKEND,), + capabilities=(Capability.TOOL_SEARCH,), + mode=Mode.NONSTREAM, + ) + ) + def test_converse_tool_result_of_only_tool_references_is_accepted( + self, proxy: ProxyClient, resources: ResourceManager, sdk: SdkClients + ) -> None: + model = _register(proxy, resources, GPT_6_1_SOL_BACKEND) + client = sdk.anthropic(resources.key()).with_options(timeout=180) + + message = client.messages.create( + model=model, + max_tokens=64, + tools=[TOOL_SEARCH, WEB_FETCH, WEB_SEARCH], + messages=[ + {"role": "user", "content": "find web tools"}, + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_1", "name": "ToolSearch", "input": {"q": "web"}}], + }, + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_1", + "content": [ + {"type": "tool_reference", "tool_name": "WebFetch"}, + {"type": "tool_reference", "tool_name": "WebSearch"}, + ], + }, + {"type": "text", "text": "reply ok"}, + ], + }, + ], # pyright: ignore[reportArgumentType] # Anthropic SDK types omit tool_reference blocks + extra_body=NO_PROXY_CACHE, + ) + + assert message.stop_reason is not None, f"Bedrock Converse returned no stop_reason: {message!r}" + assert message.content, f"Bedrock Converse returned no content blocks: {message!r}" + @meta( Subject( domain=Domain.LLM_TRANSLATION, diff --git a/tests/integration/proxy_config.yaml b/tests/integration/proxy_config.yaml index f1c0c565c54..d63a726dcfa 100644 --- a/tests/integration/proxy_config.yaml +++ b/tests/integration/proxy_config.yaml @@ -116,6 +116,12 @@ model_list: api_base: http://127.0.0.1:8191 api_key: synthetic-bedrock-key aws_region_name: us-east-1 + - model_name: bedrock/global.openai.gpt-6.1-sol + litellm_params: + model: bedrock/global.openai.gpt-6.1-sol + api_base: http://127.0.0.1:8191 + api_key: synthetic-bedrock-key + aws_region_name: us-east-1 - model_name: openai/gpt-5.4 litellm_params: model: openai/gpt-5.4 diff --git a/tests/integration/translation/messages/tool_search/test_messages_tool_search_bedrock_converse.py b/tests/integration/translation/messages/tool_search/test_messages_tool_search_bedrock_converse.py new file mode 100644 index 00000000000..2fbd4e9aa20 --- /dev/null +++ b/tests/integration/translation/messages/tool_search/test_messages_tool_search_bedrock_converse.py @@ -0,0 +1,156 @@ +from typing import Final +from unittest.mock import ANY + +import pytest +from integration._support.client import Gateway +from integration._support.provider import SharedProvider +from integration.translation.case import TranslationTestCase +from integration.translation.runner import assert_translation + +GPT_6_1_SOL_TOOL_SEARCH_TEST_CASE: Final = TranslationTestCase( + scenario="tool_search", + litellm_endpoint="/v1/messages", + litellm_request={ + "model": "bedrock/global.openai.gpt-6.1-sol", + "max_tokens": 1024, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "low"}, + "tools": [ + { + "name": "ToolSearch", + "description": "Load deferred tools by name.", + "input_schema": {"type": "object", "properties": {"query": {"type": "string"}}, "required": ["query"]}, + }, + { + "name": "WebSearch", + "description": "Search the web.", + "input_schema": {"type": "object", "properties": {"query": {"type": "string"}}, "required": ["query"]}, + }, + ], + "messages": [ + {"role": "user", "content": "Load the web search tool."}, + { + "role": "assistant", + "content": [ + { + "type": "tool_use", + "id": "call_tool_search_1", + "name": "ToolSearch", + "input": {"query": "select:WebSearch"}, + } + ], + }, + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "call_tool_search_1", + "content": [{"type": "tool_reference", "tool_name": "WebSearch"}], + } + ], + }, + ], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/model/global.openai.gpt-6.1-sol/converse", + expected_provider_headers={"authorization": "Bearer synthetic-bedrock-key", "content-type": "application/json"}, + expected_provider_request={ + "additionalModelRequestFields": {"reasoning": {"effort": "low"}}, + "inferenceConfig": {"maxTokens": 1024}, + "messages": [ + {"role": "user", "content": [{"text": "Load the web search tool."}]}, + { + "role": "assistant", + "content": [ + { + "toolUse": { + "input": {"query": "select:WebSearch"}, + "name": "ToolSearch", + "toolUseId": "call_tool_search_1", + } + } + ], + }, + { + "role": "user", + "content": [ + { + "toolResult": { + "content": [{"text": "WebSearch"}], + "toolUseId": "call_tool_search_1", + } + } + ], + }, + ], + "toolConfig": { + "tools": [ + { + "toolSpec": { + "description": "Load deferred tools by name.", + "inputSchema": { + "json": { + "type": "object", + "properties": {"query": {"type": "string"}}, + "required": ["query"], + } + }, + "name": "ToolSearch", + } + }, + { + "toolSpec": { + "description": "Search the web.", + "inputSchema": { + "json": { + "type": "object", + "properties": {"query": {"type": "string"}}, + "required": ["query"], + } + }, + "name": "WebSearch", + } + }, + ] + }, + }, + mock_provider_response={ + "metrics": {"latencyMs": 1252}, + "output": { + "message": { + "content": [{"text": "The web search tool is loaded and ready to use."}], + "role": "assistant", + } + }, + "stopReason": "end_turn", + "usage": { + "cacheReadInputTokenCount": 0, + "cacheReadInputTokens": 0, + "cacheWriteInputTokenCount": 0, + "cacheWriteInputTokens": 0, + "inputTokens": 100, + "outputTokens": 15, + "serverToolUsage": {}, + "totalTokens": 115, + }, + }, + expected_litellm_response={ + "id": ANY, + "type": "message", + "role": "assistant", + "model": "bedrock/global.openai.gpt-6.1-sol", + "stop_sequence": None, + "usage": {"input_tokens": 100, "output_tokens": 15}, + "content": [{"type": "text", "text": "The web search tool is loaded and ready to use."}], + "stop_reason": "end_turn", + "stop_details": None, + }, +) + + +@pytest.mark.parametrize("case", [GPT_6_1_SOL_TOOL_SEARCH_TEST_CASE], ids=lambda case: case.id) +def test_messages_tool_search_bedrock_converse( + case: TranslationTestCase, gateway: Gateway, provider: SharedProvider +) -> None: + assert_translation(case, gateway, provider) diff --git a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index b698f69d7c1..bb7af3f1c06 100644 --- a/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/unit/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -3264,6 +3264,41 @@ def test_convert_to_anthropic_tool_result_openai_file_pdf_becomes_document(): assert tool_result["content"][0]["document"]["source"]["bytes"] == pdf_b64 +def test_convert_to_bedrock_tool_call_result_maps_tool_references_to_text() -> None: + tool_reference_message: Final[ChatCompletionToolMessage] = { + "role": "tool", + "tool_call_id": "toolu_1", + "content": [ + {"type": "tool_reference", "tool_name": "WebFetch"}, + {"type": "tool_reference", "tool_name": "WebSearch"}, + ], + } + mixed_message: Final[ChatCompletionToolMessage] = { + "role": "tool", + "tool_call_id": "toolu_2", + "content": [ + {"type": "text", "text": "Loaded tools:"}, + {"type": "tool_reference", "tool_name": "WebSearch"}, + ], + } + + tool_reference_result: Final = _convert_to_bedrock_tool_call_result(tool_reference_message) + mixed_result: Final = _convert_to_bedrock_tool_call_result(mixed_message) + + assert tool_reference_result == { + "toolResult": { + "toolUseId": "toolu_1", + "content": [{"text": "WebFetch"}, {"text": "WebSearch"}], + } + } + assert mixed_result == { + "toolResult": { + "toolUseId": "toolu_2", + "content": [{"text": "Loaded tools:"}, {"text": "WebSearch"}], + } + } + + def test_bedrock_converse_messages_pt_document_various_formats(): """Test that various document media types produce the correct format value.""" test_cases = [