mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #18754 from BerriAI/litellm_add_annotations_responses_bridge
Add annotations to completions responses API bridge
This commit is contained in:
commit
c13bc21520
3 changed files with 232 additions and 2 deletions
|
|
@ -90,9 +90,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
content_type = content_item.get("type")
|
||||
if content_type == "output_text":
|
||||
response_text = content_item.get("text", "")
|
||||
# Extract annotations from content if present
|
||||
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
|
||||
content_item.get("annotations", None)
|
||||
)
|
||||
msg = Message(
|
||||
role=item.get("role", "assistant"),
|
||||
content=response_text if response_text else "",
|
||||
annotations=annotations,
|
||||
)
|
||||
choice = Choices(message=msg, finish_reason="stop", index=index)
|
||||
return choice, index + 1
|
||||
|
|
@ -364,10 +369,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
elif isinstance(item, ResponseOutputMessage):
|
||||
for content in item.content:
|
||||
response_text = getattr(content, "text", "")
|
||||
# Extract annotations from content if present
|
||||
raw_annotations = getattr(content, "annotations", None)
|
||||
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
|
||||
raw_annotations
|
||||
)
|
||||
msg = Message(
|
||||
role=item.role,
|
||||
content=response_text if response_text else "",
|
||||
reasoning_content=reasoning_content,
|
||||
annotations=annotations,
|
||||
)
|
||||
|
||||
choices.append(
|
||||
|
|
@ -763,6 +774,42 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
return {"format": {"type": "text"}}
|
||||
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _convert_annotations_to_chat_format(
|
||||
annotations: Optional[List[Any]],
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""
|
||||
Convert annotations from Responses API to Chat Completions format.
|
||||
|
||||
Annotations are already in compatible format between both APIs,
|
||||
so we just need to convert Pydantic models to dicts.
|
||||
"""
|
||||
if not annotations:
|
||||
return None
|
||||
|
||||
result: List[Dict[str, Any]] = []
|
||||
for annotation in annotations:
|
||||
try:
|
||||
# Convert Pydantic models to dicts (handles both v1 and v2)
|
||||
if hasattr(annotation, "model_dump"):
|
||||
annotation_dict = annotation.model_dump()
|
||||
elif hasattr(annotation, "dict"):
|
||||
annotation_dict = annotation.dict()
|
||||
elif isinstance(annotation, dict):
|
||||
annotation_dict = annotation
|
||||
else:
|
||||
# Skip unsupported annotation types
|
||||
verbose_logger.debug(f"Skipping unsupported annotation type: {type(annotation)}")
|
||||
continue
|
||||
|
||||
result.append(annotation_dict)
|
||||
except Exception as e:
|
||||
# Skip malformed annotations
|
||||
verbose_logger.debug(f"Skipping malformed annotation: {annotation}, error: {e}")
|
||||
continue
|
||||
|
||||
return result if result else None
|
||||
|
||||
def _map_responses_status_to_finish_reason(self, status: Optional[str]) -> str:
|
||||
"""Map responses API status to chat completion finish_reason"""
|
||||
|
|
|
|||
|
|
@ -336,8 +336,8 @@ class SkillsInjectionHook(CustomLogger):
|
|||
)
|
||||
|
||||
# Check if code execution is enabled for this request
|
||||
litellm_metadata = request_data.get("litellm_metadata", {})
|
||||
metadata = request_data.get("metadata", {})
|
||||
litellm_metadata = request_data.get("litellm_metadata") or {}
|
||||
metadata = request_data.get("metadata") or {}
|
||||
|
||||
code_exec_enabled = (
|
||||
litellm_metadata.get("_litellm_code_execution_enabled") or
|
||||
|
|
|
|||
|
|
@ -1095,3 +1095,186 @@ def test_map_reasoning_effort_adds_summary_detailed():
|
|||
os.environ["LITELLM_REASONING_AUTO_SUMMARY"] = original_env
|
||||
elif "LITELLM_REASONING_AUTO_SUMMARY" in os.environ:
|
||||
del os.environ["LITELLM_REASONING_AUTO_SUMMARY"]
|
||||
|
||||
|
||||
def test_transform_response_preserves_annotations():
|
||||
"""
|
||||
Test that annotations from Responses API are preserved when transforming to Chat Completions format.
|
||||
|
||||
This is a regression test for the bug where annotations (like url_citation) were being
|
||||
dropped during the transformation from ResponsesAPIResponse to ModelResponse.
|
||||
|
||||
The fix ensures annotations are extracted from ResponseOutputText content items and
|
||||
passed through to the Message object in the Chat Completions response.
|
||||
"""
|
||||
from unittest.mock import Mock
|
||||
|
||||
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
|
||||
|
||||
from litellm.completion_extras.litellm_responses_transformation.transformation import (
|
||||
LiteLLMResponsesTransformationHandler,
|
||||
)
|
||||
from litellm.types.llms.openai import (
|
||||
InputTokensDetails,
|
||||
OutputTokensDetails,
|
||||
ResponseAPIUsage,
|
||||
ResponsesAPIResponse,
|
||||
)
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
handler = LiteLLMResponsesTransformationHandler()
|
||||
|
||||
# Create annotations similar to what OpenAI Responses API returns
|
||||
annotations = [
|
||||
{
|
||||
"type": "url_citation",
|
||||
"start_index": 0,
|
||||
"end_index": 100,
|
||||
"title": "Example Article",
|
||||
"url": "https://example.com/article",
|
||||
},
|
||||
{
|
||||
"type": "url_citation",
|
||||
"start_index": 101,
|
||||
"end_index": 200,
|
||||
"title": "Another Source",
|
||||
"url": "https://example.com/source",
|
||||
},
|
||||
]
|
||||
|
||||
# Create output text with annotations
|
||||
output_text = ResponseOutputText(
|
||||
annotations=annotations,
|
||||
text="Here is some information with citations.",
|
||||
type="output_text",
|
||||
logprobs=[],
|
||||
)
|
||||
|
||||
# Create output message
|
||||
output_message = ResponseOutputMessage(
|
||||
id="msg_test123",
|
||||
content=[output_text],
|
||||
role="assistant",
|
||||
status="completed",
|
||||
type="message",
|
||||
)
|
||||
|
||||
# Create usage information
|
||||
usage = ResponseAPIUsage(
|
||||
input_tokens=10,
|
||||
input_tokens_details=InputTokensDetails(
|
||||
audio_tokens=None, cached_tokens=0, text_tokens=None
|
||||
),
|
||||
output_tokens=20,
|
||||
output_tokens_details=OutputTokensDetails(
|
||||
reasoning_tokens=0, text_tokens=None
|
||||
),
|
||||
total_tokens=30,
|
||||
cost=None,
|
||||
)
|
||||
|
||||
# Create the full ResponsesAPIResponse
|
||||
raw_response = ResponsesAPIResponse(
|
||||
id="resp_test123",
|
||||
created_at=1234567890,
|
||||
error=None,
|
||||
incomplete_details=None,
|
||||
instructions=None,
|
||||
metadata={},
|
||||
model="gpt-5.1",
|
||||
object="response",
|
||||
output=[output_message],
|
||||
parallel_tool_calls=True,
|
||||
temperature=1.0,
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
top_p=1.0,
|
||||
max_output_tokens=None,
|
||||
previous_response_id=None,
|
||||
reasoning=None,
|
||||
status="completed",
|
||||
text={"format": {"type": "text"}, "verbosity": "medium"},
|
||||
truncation="disabled",
|
||||
usage=usage,
|
||||
user=None,
|
||||
store=True,
|
||||
background=False,
|
||||
billing={"payer": "openai"},
|
||||
max_tool_calls=None,
|
||||
prompt_cache_key=None,
|
||||
safety_identifier=None,
|
||||
service_tier="default",
|
||||
top_logprobs=0,
|
||||
)
|
||||
|
||||
# Create empty model_response
|
||||
model_response = ModelResponse(
|
||||
id="chatcmpl-test123",
|
||||
created=1234567890,
|
||||
model=None,
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
choices=[],
|
||||
usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0),
|
||||
)
|
||||
|
||||
# Create mock objects for required parameters
|
||||
logging_obj = Mock()
|
||||
messages = [{"role": "user", "content": "Tell me about AI"}]
|
||||
request_data = {"model": "gpt-5.1"}
|
||||
optional_params = {}
|
||||
litellm_params = {"acompletion": False, "api_key": None}
|
||||
encoding = Mock()
|
||||
|
||||
# Call transform_response
|
||||
result = handler.transform_response(
|
||||
model="gpt-5.1",
|
||||
raw_response=raw_response,
|
||||
model_response=model_response,
|
||||
logging_obj=logging_obj,
|
||||
request_data=request_data,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
encoding=encoding,
|
||||
api_key=None,
|
||||
json_mode=None,
|
||||
)
|
||||
|
||||
# Assertions
|
||||
assert result.model == "gpt-5.1"
|
||||
assert len(result.choices) == 1
|
||||
|
||||
# Check the choice
|
||||
choice = result.choices[0]
|
||||
assert choice.finish_reason == "stop"
|
||||
assert choice.index == 0
|
||||
assert choice.message.role == "assistant"
|
||||
assert choice.message.content == "Here is some information with citations."
|
||||
|
||||
# Check that annotations are preserved
|
||||
assert hasattr(choice.message, "annotations"), "Message should have annotations attribute"
|
||||
assert choice.message.annotations is not None, "Annotations should not be None"
|
||||
assert len(choice.message.annotations) == 2, f"Expected 2 annotations, got {len(choice.message.annotations)}"
|
||||
|
||||
# Verify annotation content
|
||||
annotation1 = choice.message.annotations[0]
|
||||
assert annotation1["type"] == "url_citation"
|
||||
assert annotation1["title"] == "Example Article"
|
||||
assert annotation1["url"] == "https://example.com/article"
|
||||
assert annotation1["start_index"] == 0
|
||||
assert annotation1["end_index"] == 100
|
||||
|
||||
annotation2 = choice.message.annotations[1]
|
||||
assert annotation2["type"] == "url_citation"
|
||||
assert annotation2["title"] == "Another Source"
|
||||
assert annotation2["url"] == "https://example.com/source"
|
||||
assert annotation2["start_index"] == 101
|
||||
assert annotation2["end_index"] == 200
|
||||
|
||||
# Check usage
|
||||
assert result.usage.prompt_tokens == 10
|
||||
assert result.usage.completion_tokens == 20
|
||||
assert result.usage.total_tokens == 30
|
||||
|
||||
print("✓ Annotations from Responses API are correctly preserved in Chat Completions format")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue