diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index 26e6016095e..fb5fcb3efc5 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -719,19 +719,28 @@ class ModelResponseIterator: content_block_start=content_block_start, provider_specific_fields=provider_specific_fields, ) - elif ( - content_block_start["content_block"]["type"] - == "web_search_tool_result" - ): - # Capture web_search_tool_result for multi-turn reconstruction - # The full content comes in content_block_start, not in deltas - # See: https://github.com/BerriAI/litellm/issues/17737 - self.web_search_results.append( - content_block_start["content_block"] - ) - provider_specific_fields["web_search_results"] = ( - self.web_search_results - ) + elif content_block_start["content_block"]["type"].endswith("_tool_result"): + # Handle all tool result types (web_search, bash_code_execution, text_editor, etc.) + content_type = content_block_start["content_block"]["type"] + + # Special handling for web_search_tool_result for backwards compatibility + if content_type == "web_search_tool_result": + # Capture web_search_tool_result for multi-turn reconstruction + # The full content comes in content_block_start, not in deltas + # See: https://github.com/BerriAI/litellm/issues/17737 + self.web_search_results.append( + content_block_start["content_block"] + ) + provider_specific_fields["web_search_results"] = ( + self.web_search_results + ) + elif content_type != "tool_search_tool_result": + # Handle other tool results (code execution, etc.) + # Skip tool_search_tool_result as it's internal metadata + if not hasattr(self, "tool_results"): + self.tool_results = [] + self.tool_results.append(content_block_start["content_block"]) + provider_specific_fields["tool_results"] = self.tool_results elif type_chunk == "content_block_stop": ContentBlockStop(**chunk) # type: ignore # check if tool call content block - only for tool_use and server_tool_use blocks diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 57391c152cb..3a66797c67e 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1131,6 +1131,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): Optional[str], List[ChatCompletionToolCallChunk], Optional[List[Any]], + Optional[List[Any]], ]: text_content = "" citations: Optional[List[Any]] = None @@ -1142,6 +1143,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): reasoning_content: Optional[str] = None tool_calls: List[ChatCompletionToolCallChunk] = [] web_search_results: Optional[List[Any]] = None + tool_results: Optional[List[Any]] = None for idx, content in enumerate(completion_response["content"]): if content["type"] == "text": text_content += content["text"] @@ -1152,16 +1154,21 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): index=idx, ) tool_calls.append(tool_call) - ## TOOL SEARCH TOOL RESULT (skip - this is metadata about tool discovery) - elif content["type"] == "tool_search_tool_result": - # This block contains tool_references that were discovered - # We don't need to include this in the response as it's internal metadata - pass - ## WEB SEARCH TOOL RESULT - preserve web search results for multi-turn conversations - elif content["type"] == "web_search_tool_result": - if web_search_results is None: - web_search_results = [] - web_search_results.append(content) + ## TOOL RESULTS - handle all tool result types (code execution, etc.) + elif content["type"].endswith("_tool_result"): + # Skip tool_search_tool_result as it's internal metadata + if content["type"] == "tool_search_tool_result": + continue + # Handle web_search_tool_result separately for backwards compatibility + if content["type"] == "web_search_tool_result": + if web_search_results is None: + web_search_results = [] + web_search_results.append(content) + else: + # All other tool results (bash_code_execution_tool_result, text_editor_code_execution_tool_result, etc.) + if tool_results is None: + tool_results = [] + tool_results.append(content) elif content.get("thinking", None) is not None: if thinking_blocks is None: thinking_blocks = [] @@ -1193,7 +1200,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): if thinking_content is not None: reasoning_content += thinking_content - return text_content, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results + return text_content, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results def calculate_usage( self, @@ -1335,6 +1342,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): reasoning_content, tool_calls, web_search_results, + tool_results, ) = self.extract_response_content(completion_response=completion_response) if ( @@ -1358,6 +1366,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): provider_specific_fields["context_management"] = context_management if web_search_results is not None: provider_specific_fields["web_search_results"] = web_search_results + if tool_results is not None: + provider_specific_fields["tool_results"] = tool_results if container is not None: provider_specific_fields["container"] = container diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 5e58601d589..6fca423142d 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1812,3 +1812,242 @@ def test_calculate_usage_completion_tokens_details_with_reasoning(): expected_text_tokens = 500 - usage.completion_tokens_details.reasoning_tokens assert usage.completion_tokens_details.text_tokens == expected_text_tokens assert usage.completion_tokens == 500 + + +def test_code_execution_tool_results_extraction(): + """ + Test that code execution tool results (bash_code_execution_tool_result, + text_editor_code_execution_tool_result) are properly extracted and exposed + in provider_specific_fields. + + Related to: https://github.com/BerriAI/litellm/issues/xxxxx + """ + import httpx + from litellm.types.utils import ModelResponse + + config = AnthropicConfig() + + # Mock Anthropic response with code execution tool results + mock_anthropic_response = { + "id": "msg_01XYZ", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [ + { + "type": "text", + "text": "I'll calculate that for you." + }, + { + "type": "server_tool_use", + "id": "srvtoolu_01ABC", + "name": "bash_code_execution", + "input": { + "command": "python3 << 'EOF'\nprint(2 + 2)\nEOF\n" + } + }, + { + "type": "bash_code_execution_tool_result", + "tool_use_id": "srvtoolu_01ABC", + "content": { + "type": "bash_code_execution_result", + "stdout": "4\n", + "stderr": "", + "return_code": 0 + } + }, + { + "type": "server_tool_use", + "id": "srvtoolu_01DEF", + "name": "text_editor_code_execution", + "input": { + "command": "create", + "path": "test.txt", + "file_text": "Hello" + } + }, + { + "type": "text_editor_code_execution_tool_result", + "tool_use_id": "srvtoolu_01DEF", + "content": { + "type": "text_editor_code_execution_result", + "is_file_update": False + } + }, + { + "type": "text", + "text": "Done!" + } + ], + "stop_reason": "stop", + "stop_sequence": None, + "usage": { + "input_tokens": 100, + "output_tokens": 50 + } + } + + # Create mock HTTP response + mock_raw_response = MagicMock(spec=httpx.Response) + mock_raw_response.json.return_value = mock_anthropic_response + mock_raw_response.status_code = 200 + mock_raw_response.headers = {} + + model_response = ModelResponse() + + transformed_response = config.transform_parsed_response( + completion_response=mock_anthropic_response, + raw_response=mock_raw_response, + model_response=model_response, + json_mode=False, + prefix_prompt=None, + ) + + # Verify tool calls are present + assert transformed_response.choices[0].message.tool_calls is not None + assert len(transformed_response.choices[0].message.tool_calls) == 2 + + # Verify first tool call + assert transformed_response.choices[0].message.tool_calls[0].id == "srvtoolu_01ABC" + assert transformed_response.choices[0].message.tool_calls[0].function.name == "bash_code_execution" + + # Verify second tool call + assert transformed_response.choices[0].message.tool_calls[1].id == "srvtoolu_01DEF" + assert transformed_response.choices[0].message.tool_calls[1].function.name == "text_editor_code_execution" + + # Verify tool results are in provider_specific_fields + provider_fields = transformed_response.choices[0].message.provider_specific_fields + assert provider_fields is not None + assert "tool_results" in provider_fields + assert provider_fields["tool_results"] is not None + assert len(provider_fields["tool_results"]) == 2 + + # Verify bash_code_execution_tool_result + bash_result = provider_fields["tool_results"][0] + assert bash_result["type"] == "bash_code_execution_tool_result" + assert bash_result["tool_use_id"] == "srvtoolu_01ABC" + assert bash_result["content"]["stdout"] == "4\n" + assert bash_result["content"]["return_code"] == 0 + + # Verify text_editor_code_execution_tool_result + editor_result = provider_fields["tool_results"][1] + assert editor_result["type"] == "text_editor_code_execution_tool_result" + assert editor_result["tool_use_id"] == "srvtoolu_01DEF" + assert editor_result["content"]["is_file_update"] is False + + # Verify text content is properly concatenated + assert "I'll calculate that for you." in transformed_response.choices[0].message.content + assert "Done!" in transformed_response.choices[0].message.content + + +def test_tool_search_tool_result_not_in_tool_results(): + """ + Test that tool_search_tool_result is NOT included in tool_results + since it's internal metadata, not actual tool execution results. + """ + import httpx + from litellm.types.utils import ModelResponse + + config = AnthropicConfig() + + mock_anthropic_response = { + "id": "msg_01XYZ", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [ + { + "type": "text", + "text": "Found tools." + }, + { + "type": "tool_search_tool_result", + "tool_references": ["tool1", "tool2"] + } + ], + "stop_reason": "stop", + "stop_sequence": None, + "usage": { + "input_tokens": 100, + "output_tokens": 50 + } + } + + mock_raw_response = MagicMock(spec=httpx.Response) + mock_raw_response.json.return_value = mock_anthropic_response + mock_raw_response.status_code = 200 + mock_raw_response.headers = {} + + model_response = ModelResponse() + + transformed_response = config.transform_parsed_response( + completion_response=mock_anthropic_response, + raw_response=mock_raw_response, + model_response=model_response, + json_mode=False, + prefix_prompt=None, + ) + + # Verify tool_search_tool_result is NOT in tool_results + provider_fields = transformed_response.choices[0].message.provider_specific_fields + assert provider_fields.get("tool_results") is None + + +def test_web_search_tool_result_backwards_compatibility(): + """ + Test that web_search_tool_result continues to be stored in web_search_results + for backwards compatibility, not in tool_results. + """ + import httpx + from litellm.types.utils import ModelResponse + + config = AnthropicConfig() + + mock_anthropic_response = { + "id": "msg_01XYZ", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [ + { + "type": "text", + "text": "Here are the results." + }, + { + "type": "web_search_tool_result", + "search_query": "test query", + "results": [{"title": "Result 1", "url": "https://example.com"}] + } + ], + "stop_reason": "stop", + "stop_sequence": None, + "usage": { + "input_tokens": 100, + "output_tokens": 50 + } + } + + mock_raw_response = MagicMock(spec=httpx.Response) + mock_raw_response.json.return_value = mock_anthropic_response + mock_raw_response.status_code = 200 + mock_raw_response.headers = {} + + model_response = ModelResponse() + + transformed_response = config.transform_parsed_response( + completion_response=mock_anthropic_response, + raw_response=mock_raw_response, + model_response=model_response, + json_mode=False, + prefix_prompt=None, + ) + + # Verify web_search_tool_result is in web_search_results (not tool_results) + provider_fields = transformed_response.choices[0].message.provider_specific_fields + assert "web_search_results" in provider_fields + assert provider_fields["web_search_results"] is not None + assert len(provider_fields["web_search_results"]) == 1 + assert provider_fields["web_search_results"][0]["type"] == "web_search_tool_result" + + # Should NOT be in tool_results + assert provider_fields.get("tool_results") is None