feat(messages): add web search response translation for native /v1/messages handler

Map OpenAI Responses API web_search_call events to Anthropic
server_tool_use + web_search_tool_result + cited text blocks,
for both streaming (SSE) and non-streaming paths.

Handles search, open_page, and find_in_page action types.
Emits a single aggregated web_search_tool_result per response.
Adds server_tool_use counts (web_search_requests, web_fetch_requests)
to the usage object.
This commit is contained in:
Vigilans 2026-04-19 20:46:34 +08:00
parent 4148667671
commit 21b9c64f19
3 changed files with 306 additions and 1 deletions

View file

@ -7,6 +7,11 @@ from typing import Any, AsyncIterator, Dict
from litellm import verbose_logger
from litellm._uuid import uuid
from .utils import (
build_text_blocks_with_citations,
build_web_tool_use,
build_web_search_results_from_annotations,
)
class AnthropicResponsesStreamWrapper:
@ -41,6 +46,8 @@ class AnthropicResponsesStreamWrapper:
self._sent_message_start = False
self._sent_message_stop = False
self._chunk_queue: deque = deque()
# Web tools: defer text emission until output_item.done
self._web_tool_uses: list = []
def _make_message_start(self) -> Dict[str, Any]:
return {
@ -96,6 +103,10 @@ class AnthropicResponsesStreamWrapper:
)
if item_type == "message":
if self._web_tool_uses:
# Don't open text block here — deferred to
# output_item.done where full annotations (citations) are available.
return
block_idx = self._next_block_index()
if item_id:
self._item_id_to_block_index[item_id] = block_idx
@ -133,6 +144,10 @@ class AnthropicResponsesStreamWrapper:
},
}
)
elif item_type == "web_search_call":
# Don't emit content_block_start here — search queries
# are only available in output_item.done.
pass
elif item_type == "reasoning":
block_idx = self._next_block_index()
if item_id:
@ -148,6 +163,10 @@ class AnthropicResponsesStreamWrapper:
# ---- text delta ----
if event_type == "response.output_text.delta":
if self._web_tool_uses:
# Don't stream text deltas here — full text with citation
# is emitted from output_item.done.
return
item_id = getattr(event, "item_id", None) or (
event.get("item_id") if isinstance(event, dict) else None
)
@ -223,6 +242,29 @@ class AnthropicResponsesStreamWrapper:
if item
else None
)
item_type = (
getattr(item, "type", None)
or (item.get("type") if isinstance(item, dict) else None)
if item
else None
)
if item_type == "reasoning":
# Reasoning items are closed by individual part.done events
return
if item_type == "web_search_call":
self._emit_web_tool_use(item)
return
if item_type == "message" and self._web_tool_uses:
if not isinstance(item, dict):
item = item.model_dump()
for part in item.get("content") or []:
if isinstance(part, dict) and part.get("type") == "output_text":
text = part.get("text", "")
annotations = part.get("annotations") or []
citations = self._emit_web_search_results(annotations)
self._emit_cited_text_blocks(text, citations)
break
return
block_idx = (
self._item_id_to_block_index.get(item_id, self._current_block_index)
if item_id
@ -288,6 +330,11 @@ class AnthropicResponsesStreamWrapper:
usage_delta["cache_creation_input_tokens"] = cache_creation_tokens
if cache_read_tokens:
usage_delta["cache_read_input_tokens"] = cache_read_tokens
if self._web_tool_uses:
usage_delta["server_tool_use"] = {
"web_search_requests": sum(c["name"] == "web_search" for c in self._web_tool_uses),
"web_fetch_requests": sum(c["name"] == "web_fetch" for c in self._web_tool_uses),
}
self._chunk_queue.append(
{
@ -300,6 +347,81 @@ class AnthropicResponsesStreamWrapper:
self._sent_message_stop = True
return
def _emit_web_tool_use(self, item: Any) -> None:
"""Emit server_tool_use block and collect web tool info."""
block, input_dict = build_web_tool_use(item)
block_idx = self._next_block_index()
self._web_tool_uses.append(block)
self._chunk_queue.append(
{
"type": "content_block_start",
"index": block_idx,
"content_block": block,
}
)
self._chunk_queue.append(
{
"type": "content_block_delta",
"index": block_idx,
"delta": {
"type": "input_json_delta",
"partial_json": json.dumps(input_dict),
},
}
)
self._chunk_queue.append(
{"type": "content_block_stop", "index": block_idx}
)
def _emit_web_search_results(self, annotations: list) -> list:
"""Emit web_search_tool_result blocks and return citations for text emission."""
blocks, citations = build_web_search_results_from_annotations(
self._web_tool_uses, annotations
)
for block in blocks:
block_idx = self._next_block_index()
self._chunk_queue.append(
{
"type": "content_block_start",
"index": block_idx,
"content_block": block,
}
)
self._chunk_queue.append(
{"type": "content_block_stop", "index": block_idx}
)
return citations
def _emit_cited_text_blocks(self, text: str, citations: list) -> None:
"""Emit text blocks with citation deltas."""
for block_data in build_text_blocks_with_citations(text, citations):
block_idx = self._next_block_index()
self._chunk_queue.append(
{
"type": "content_block_start",
"index": block_idx,
"content_block": {"type": "text", "text": ""},
}
)
self._chunk_queue.append(
{
"type": "content_block_delta",
"index": block_idx,
"delta": {"type": "text_delta", "text": block_data["text"]},
}
)
for cit in block_data.get("citations", []):
self._chunk_queue.append(
{
"type": "content_block_delta",
"index": block_idx,
"delta": {"type": "citations_delta", "citation": cit},
}
)
self._chunk_queue.append(
{"type": "content_block_stop", "index": block_idx}
)
def __aiter__(self) -> "AnthropicResponsesStreamWrapper":
return self

View file

@ -11,6 +11,12 @@ from typing import Any, Dict, List, Optional, Union, cast
from litellm.llms.anthropic.experimental_pass_through.utils import (
is_reasoning_auto_summary_enabled,
)
from .utils import (
build_text_blocks_with_citations,
build_web_tool_use,
build_web_search_results_from_annotations,
)
from litellm.types.llms.anthropic import (
AllAnthropicToolsValues,
AnthopicMessagesAssistantMessageParam,
@ -421,6 +427,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
"""
from openai.types.responses import (
ResponseFunctionToolCall,
ResponseFunctionWebSearch,
ResponseOutputMessage,
ResponseReasoningItem,
)
@ -429,6 +436,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
content: List[Dict[str, Any]] = []
stop_reason: AnthropicFinishReason = "end_turn"
web_tool_uses: List[Dict[str, Any]] = []
for item in response.output:
if isinstance(item, ResponseReasoningItem):
@ -443,6 +451,23 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
).model_dump()
)
elif isinstance(item, ResponseFunctionWebSearch):
block, input_dict = build_web_tool_use(item)
web_tool_uses.append(block)
content.append({**block, "input": input_dict})
elif isinstance(item, ResponseOutputMessage) and web_tool_uses:
for part in item.content:
if getattr(part, "type", None) == "output_text":
blocks, citations = build_web_search_results_from_annotations(
web_tool_uses, getattr(part, "annotations", []) or []
)
content.extend(blocks)
content.extend(
build_text_blocks_with_citations(getattr(part, "text", ""), citations)
)
break
elif isinstance(item, ResponseOutputMessage):
for part in item.content:
if getattr(part, "type", None) == "output_text":
@ -469,7 +494,22 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
elif isinstance(item, dict):
item_type = item.get("type")
if item_type == "message":
if item_type == "web_search_call":
block, input_dict = build_web_tool_use(item)
web_tool_uses.append(block)
content.append({**block, "input": input_dict})
elif item_type == "message" and web_tool_uses:
for part in item.get("content", []):
if isinstance(part, dict) and part.get("type") == "output_text":
blocks, citations = build_web_search_results_from_annotations(
web_tool_uses, part.get("annotations") or []
)
content.extend(blocks)
content.extend(
build_text_blocks_with_citations(part.get("text", ""), citations)
)
break
elif item_type == "message":
for part in item.get("content", []):
if isinstance(part, dict) and part.get("type") == "output_text":
content.append(
@ -506,6 +546,12 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
output_tokens=output_tokens,
)
if web_tool_uses:
anthropic_usage["server_tool_use"] = { # type: ignore[typeddict-unknown-key]
"web_search_requests": sum(c["name"] == "web_search" for c in web_tool_uses),
"web_fetch_requests": sum(c["name"] == "web_fetch" for c in web_tool_uses),
}
return AnthropicMessagesResponse(
id=response.id,
type="message",

View file

@ -0,0 +1,137 @@
"""
Shared utilities for Anthropic <-> OpenAI Responses API web search translation.
Used by both the streaming (streaming_iterator.py) and non-streaming
(transformation.py) response paths.
"""
from typing import Any, Dict, List, Tuple
from litellm.types.llms.anthropic import AnthropicResponseContentBlockText
def build_web_tool_use(item: Any) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""Build server_tool_use content block from a response web_search_call item.
Returns (server_tool_use_block, input_dict) where:
- server_tool_use_block: {"type": "server_tool_use", "id": ..., "name": "web_search"|"web_fetch"}
- input_dict: {"query": "..."} for search, {"url": "..."} for open_page/find_in_page
"""
if not isinstance(item, dict):
item = item.model_dump()
action = item.get("action", {})
action_type = action.get("type", "search")
if action_type in ("open_page", "find_in_page"):
name = "web_fetch"
input_dict = {"url": action.get("url", "")}
else:
name = "web_search"
queries = action.get("queries", {})
if queries:
query = "\n".join(queries)
else:
query = action.get("query", "")
input_dict = {"query": query}
block = {
"type": "server_tool_use",
"name": name,
"id": item.get("id", ""),
}
return block, input_dict
def build_web_search_results_from_annotations(
web_tool_uses: List[Dict[str, Any]],
annotations: list,
) -> Tuple[List[Dict[str, Any]], List[tuple]]:
"""Build web search content blocks and citations from search calls and annotations.
Returns (content_blocks, citations) where:
- content_blocks: web_search_tool_result blocks
- citations: list of (start_index, end_index, citation_dict) tuples sorted by position
"""
citations: List[tuple] = []
seen_urls: Dict[str, Dict[str, Any]] = {}
for ann in annotations:
if not isinstance(ann, dict):
ann = ann.model_dump()
if ann.get("type") != "url_citation":
continue
url = ann.get("url", "")
title = ann.get("title", "")
start = ann.get("start_index", 0) or 0
end = ann.get("end_index", 0) or 0
citations.append((start, end, {
"type": "web_search_result_location",
"url": url,
"title": title,
"cited_text": title,
}))
if url and url not in seen_urls:
seen_urls[url] = {
"type": "web_search_result",
"url": url,
"title": title,
}
citations.sort(key=lambda x: x[0])
search_results = list(seen_urls.values())
search_calls = [c for c in web_tool_uses if c["name"] == "web_search"]
content_blocks: List[Dict[str, Any]] = []
if search_calls:
content_blocks.append(
{
"type": "web_search_tool_result",
"tool_use_id": search_calls[0]["id"],
"content": search_results,
}
)
return content_blocks, citations
def build_text_blocks_with_citations(
text: str,
citations: List[tuple],
) -> List[Dict[str, Any]]:
"""Split text into alternating uncited / cited Anthropic text blocks.
Each citation tuple is (start_index, end_index, citation_dict).
text[start:end] is the cited range; everything else is uncited.
"""
if not citations:
return [
AnthropicResponseContentBlockText(type="text", text=text).model_dump()
]
blocks: List[Dict[str, Any]] = []
pos = 0
for start, end, citation in citations:
if pos < start:
blocks.append(
AnthropicResponseContentBlockText(
type="text", text=text[pos:start]
).model_dump()
)
if start < end:
block = AnthropicResponseContentBlockText(
type="text", text=text[start:end]
).model_dump()
block["citations"] = [citation]
blocks.append(block)
pos = end
if pos < len(text):
blocks.append(
AnthropicResponseContentBlockText(
type="text", text=text[pos:]
).model_dump()
)
return blocks