Merge pull request #18945 from BerriAI/litellm_add_anthropic_tool_call_results

Add: missing anthropic tool results in response
This commit is contained in:
Sameer Kankute 2026-01-12 22:11:51 +05:30 • committed by GitHub
commit c2fcc6aa92
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 309 additions and 49 deletions

View file

@ -719,32 +719,40 @@ class ModelResponseIterator:
content_block_start=content_block_start,
provider_specific_fields=provider_specific_fields,
)
elif (
content_block_start["content_block"]["type"]
== "web_search_tool_result"
):
# Capture web_search_tool_result for multi-turn reconstruction
# The full content comes in content_block_start, not in deltas
# See: https://github.com/BerriAI/litellm/issues/17737
self.web_search_results.append(
content_block_start["content_block"]
)
provider_specific_fields["web_search_results"] = (
self.web_search_results
)
elif (
content_block_start["content_block"]["type"]
== "web_fetch_tool_result"
):
# Capture web_fetch_tool_result for multi-turn reconstruction
# The full content comes in content_block_start, not in deltas
# Fixes: https://github.com/BerriAI/litellm/issues/18137
self.web_search_results.append(
content_block_start["content_block"]
)
provider_specific_fields["web_search_results"] = (
self.web_search_results
)
elif content_block_start["content_block"]["type"].endswith("_tool_result"):
# Handle all tool result types (web_search, bash_code_execution, text_editor, etc.)
content_type = content_block_start["content_block"]["type"]
# Special handling for web_search_tool_result for backwards compatibility
if content_type == "web_search_tool_result":
# Capture web_search_tool_result for multi-turn reconstruction
# The full content comes in content_block_start, not in deltas
# See: https://github.com/BerriAI/litellm/issues/17737
self.web_search_results.append(
content_block_start["content_block"]
)
provider_specific_fields["web_search_results"] = (
self.web_search_results
)
elif content_type == "web_fetch_tool_result":
# Capture web_fetch_tool_result for multi-turn reconstruction
# The full content comes in content_block_start, not in deltas
# Fixes: https://github.com/BerriAI/litellm/issues/18137
self.web_search_results.append(
content_block_start["content_block"]
)
provider_specific_fields["web_search_results"] = (
self.web_search_results
)
elif content_type != "tool_search_tool_result":
# Handle other tool results (code execution, etc.)
# Skip tool_search_tool_result as it's internal metadata
if not hasattr(self, "tool_results"):
self.tool_results = []
self.tool_results.append(content_block_start["content_block"])
provider_specific_fields["tool_results"] = self.tool_results
elif type_chunk == "content_block_stop":
ContentBlockStop(**chunk) # type: ignore
# check if tool call content block - only for tool_use and server_tool_use blocks

View file

@ -1138,6 +1138,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
Optional[str],
List[ChatCompletionToolCallChunk],
Optional[List[Any]],
Optional[List[Any]],
]:
text_content = ""
citations: Optional[List[Any]] = None
@ -1149,6 +1150,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
reasoning_content: Optional[str] = None
tool_calls: List[ChatCompletionToolCallChunk] = []
web_search_results: Optional[List[Any]] = None
tool_results: Optional[List[Any]] = None
for idx, content in enumerate(completion_response["content"]):
if content["type"] == "text":
text_content += content["text"]
@ -1159,22 +1161,27 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
index=idx,
)
tool_calls.append(tool_call)
## TOOL SEARCH TOOL RESULT (skip - this is metadata about tool discovery)
elif content["type"] == "tool_search_tool_result":
# This block contains tool_references that were discovered
# We don't need to include this in the response as it's internal metadata
pass
## WEB SEARCH TOOL RESULT - preserve web search results for multi-turn conversations
elif content["type"] == "web_search_tool_result":
if web_search_results is None:
web_search_results = []
web_search_results.append(content)
## WEB FETCH TOOL RESULT - preserve web fetch results for multi-turn conversations
## Fixes: https://github.com/BerriAI/litellm/issues/18137
elif content["type"] == "web_fetch_tool_result":
if web_search_results is None:
web_search_results = []
web_search_results.append(content)
## TOOL RESULTS - handle all tool result types (code execution, etc.)
elif content["type"].endswith("_tool_result"):
# Skip tool_search_tool_result as it's internal metadata
if content["type"] == "tool_search_tool_result":
continue
# Handle web_search_tool_result separately for backwards compatibility
if content["type"] == "web_search_tool_result":
if web_search_results is None:
web_search_results = []
web_search_results.append(content)
elif content["type"] == "web_fetch_tool_result":
if web_search_results is None:
web_search_results = []
web_search_results.append(content)
else:
# All other tool results (bash_code_execution_tool_result, text_editor_code_execution_tool_result, etc.)
if tool_results is None:
tool_results = []
tool_results.append(content)
elif content.get("thinking", None) is not None:
if thinking_blocks is None:
thinking_blocks = []
@ -1206,7 +1213,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
if thinking_content is not None:
reasoning_content += thinking_content
return text_content, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results
return text_content, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results
def calculate_usage(
self,
@ -1348,6 +1355,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
reasoning_content,
tool_calls,
web_search_results,
tool_results,
) = self.extract_response_content(completion_response=completion_response)
if (
@ -1371,6 +1379,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
provider_specific_fields["context_management"] = context_management
if web_search_results is not None:
provider_specific_fields["web_search_results"] = web_search_results
if tool_results is not None:
provider_specific_fields["tool_results"] = tool_results
if container is not None:
provider_specific_fields["container"] = container

View file

@ -185,7 +185,7 @@ def test_extract_response_content_with_citations():
},
}
_, citations, _, _, _, _ = config.extract_response_content(completion_response)
_, citations, _, _, _, _ , _= config.extract_response_content(completion_response)
assert citations == [
[
{
@ -342,7 +342,7 @@ def test_web_search_tool_result_extraction():
}
}
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results = config.extract_response_content(
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results = config.extract_response_content(
completion_response
)
@ -474,7 +474,7 @@ def test_multiple_web_search_tool_results():
]
}
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results = config.extract_response_content(
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results = config.extract_response_content(
completion_response
)
@ -923,7 +923,7 @@ def test_server_tool_use_in_response():
]
}
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results = config.extract_response_content(
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results = config.extract_response_content(
completion_response
)
@ -1051,7 +1051,7 @@ def test_tool_search_complete_response_parsing():
}
# Extract content
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results = config.extract_response_content(
text, citations, thinking_blocks, reasoning_content, tool_calls, web_search_results, tool_results = config.extract_response_content(
completion_response
)
@ -1171,7 +1171,7 @@ def test_caller_field_in_response():
"usage": {"input_tokens": 100, "output_tokens": 50}
}
text, citations, thinking, reasoning, tool_calls, web_search_results = config.extract_response_content(completion_response)
text, citations, thinking, reasoning, tool_calls, web_search_results, tool_results = config.extract_response_content(completion_response)
assert len(tool_calls) == 1
assert tool_calls[0]["id"] == "toolu_123"
@ -1812,3 +1812,245 @@ def test_calculate_usage_completion_tokens_details_with_reasoning():
expected_text_tokens = 500 - usage.completion_tokens_details.reasoning_tokens
assert usage.completion_tokens_details.text_tokens == expected_text_tokens
assert usage.completion_tokens == 500
def test_code_execution_tool_results_extraction():
"""
Test that code execution tool results (bash_code_execution_tool_result,
text_editor_code_execution_tool_result) are properly extracted and exposed
in provider_specific_fields.
Related to: https://github.com/BerriAI/litellm/issues/xxxxx
"""
import httpx
from litellm.types.utils import ModelResponse
config = AnthropicConfig()
# Mock Anthropic response with code execution tool results
mock_anthropic_response = {
"id": "msg_01XYZ",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5-20250929",
"content": [
{
"type": "text",
"text": "I'll calculate that for you."
},
{
"type": "server_tool_use",
"id": "srvtoolu_01ABC",
"name": "bash_code_execution",
"input": {
"command": "python3 << 'EOF'\nprint(2 + 2)\nEOF\n"
}
},
{
"type": "bash_code_execution_tool_result",
"tool_use_id": "srvtoolu_01ABC",
"content": {
"type": "bash_code_execution_result",
"stdout": "4\n",
"stderr": "",
"return_code": 0
}
},
{
"type": "server_tool_use",
"id": "srvtoolu_01DEF",
"name": "text_editor_code_execution",
"input": {
"command": "create",
"path": "test.txt",
"file_text": "Hello"
}
},
{
"type": "text_editor_code_execution_tool_result",
"tool_use_id": "srvtoolu_01DEF",
"content": {
"type": "text_editor_code_execution_result",
"is_file_update": False
}
},
{
"type": "text",
"text": "Done!"
}
],
"stop_reason": "stop",
"stop_sequence": None,
"usage": {
"input_tokens": 100,
"output_tokens": 50
}
}
# Create mock HTTP response
mock_raw_response = MagicMock(spec=httpx.Response)
mock_raw_response.json.return_value = mock_anthropic_response
mock_raw_response.status_code = 200
mock_raw_response.headers = {}
model_response = ModelResponse()
transformed_response = config.transform_parsed_response(
completion_response=mock_anthropic_response,
raw_response=mock_raw_response,
model_response=model_response,
json_mode=False,
prefix_prompt=None,
)
# Verify tool calls are present
assert transformed_response.choices[0].message.tool_calls is not None
assert len(transformed_response.choices[0].message.tool_calls) == 2
# Verify first tool call
assert transformed_response.choices[0].message.tool_calls[0].id == "srvtoolu_01ABC"
assert transformed_response.choices[0].message.tool_calls[0].function.name == "bash_code_execution"
# Verify second tool call
assert transformed_response.choices[0].message.tool_calls[1].id == "srvtoolu_01DEF"
assert transformed_response.choices[0].message.tool_calls[1].function.name == "text_editor_code_execution"
# Verify tool results are in provider_specific_fields
provider_fields = transformed_response.choices[0].message.provider_specific_fields
assert provider_fields is not None
assert "tool_results" in provider_fields
assert provider_fields["tool_results"] is not None
assert len(provider_fields["tool_results"]) == 2
# Verify bash_code_execution_tool_result
bash_result = provider_fields["tool_results"][0]
assert bash_result["type"] == "bash_code_execution_tool_result"
assert bash_result["tool_use_id"] == "srvtoolu_01ABC"
assert bash_result["content"]["stdout"] == "4\n"
assert bash_result["content"]["return_code"] == 0
# Verify text_editor_code_execution_tool_result
editor_result = provider_fields["tool_results"][1]
assert editor_result["type"] == "text_editor_code_execution_tool_result"
assert editor_result["tool_use_id"] == "srvtoolu_01DEF"
assert editor_result["content"]["is_file_update"] is False
# Verify text content is properly concatenated
assert "I'll calculate that for you." in transformed_response.choices[0].message.content
assert "Done!" in transformed_response.choices[0].message.content
def test_tool_search_tool_result_not_in_tool_results():
"""
Test that tool_search_tool_result is NOT included in tool_results
since it's internal metadata, not actual tool execution results.
"""
import httpx
from litellm.types.utils import ModelResponse
config = AnthropicConfig()
mock_anthropic_response = {
"id": "msg_01XYZ",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5-20250929",
"content": [
{
"type": "text",
"text": "Found tools."
},
{
"type": "tool_search_tool_result",
"tool_references": ["tool1", "tool2"]
}
],
"stop_reason": "stop",
"stop_sequence": None,
"usage": {
"input_tokens": 100,
"output_tokens": 50
}
}
mock_raw_response = MagicMock(spec=httpx.Response)
mock_raw_response.json.return_value = mock_anthropic_response
mock_raw_response.status_code = 200
mock_raw_response.headers = {}
model_response = ModelResponse()
transformed_response = config.transform_parsed_response(
completion_response=mock_anthropic_response,
raw_response=mock_raw_response,
model_response=model_response,
json_mode=False,
prefix_prompt=None,
)
# Verify tool_search_tool_result is NOT in tool_results
provider_fields = transformed_response.choices[0].message.provider_specific_fields
assert provider_fields.get("tool_results") is None
def test_web_search_tool_result_backwards_compatibility():
"""
Test that web_search_tool_result continues to be stored in web_search_results
for backwards compatibility, not in tool_results.
"""
import httpx
from litellm.types.utils import ModelResponse
config = AnthropicConfig()
mock_anthropic_response = {
"id": "msg_01XYZ",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5-20250929",
"content": [
{
"type": "text",
"text": "Here are the results."
},
{
"type": "web_search_tool_result",
"search_query": "test query",
"results": [{"title": "Result 1", "url": "https://example.com"}]
}
],
"stop_reason": "stop",
"stop_sequence": None,
"usage": {
"input_tokens": 100,
"output_tokens": 50
}
}
mock_raw_response = MagicMock(spec=httpx.Response)
mock_raw_response.json.return_value = mock_anthropic_response
mock_raw_response.status_code = 200
mock_raw_response.headers = {}
model_response = ModelResponse()
transformed_response = config.transform_parsed_response(
completion_response=mock_anthropic_response,
raw_response=mock_raw_response,
model_response=model_response,
json_mode=False,
prefix_prompt=None,
)
# Verify web_search_tool_result is in web_search_results (not tool_results)
provider_fields = transformed_response.choices[0].message.provider_specific_fields
assert "web_search_results" in provider_fields
assert provider_fields["web_search_results"] is not None
assert len(provider_fields["web_search_results"]) == 1
assert provider_fields["web_search_results"][0]["type"] == "web_search_tool_result"
# Should NOT be in tool_results
assert provider_fields.get("tool_results") is None