Merge pull request #18754 from BerriAI/litellm_add_annotations_responses_bridge

Add annotations to completions responses API bridge
This commit is contained in:
Sameer Kankute 2026-01-08 15:30:16 +05:30 • committed by GitHub
commit c13bc21520
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 232 additions and 2 deletions

View file

@ -90,9 +90,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
content_type = content_item.get("type")
if content_type == "output_text":
response_text = content_item.get("text", "")
# Extract annotations from content if present
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
content_item.get("annotations", None)
)
msg = Message(
role=item.get("role", "assistant"),
content=response_text if response_text else "",
annotations=annotations,
)
choice = Choices(message=msg, finish_reason="stop", index=index)
return choice, index + 1
@ -364,10 +369,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
elif isinstance(item, ResponseOutputMessage):
for content in item.content:
response_text = getattr(content, "text", "")
# Extract annotations from content if present
raw_annotations = getattr(content, "annotations", None)
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
raw_annotations
)
msg = Message(
role=item.role,
content=response_text if response_text else "",
reasoning_content=reasoning_content,
annotations=annotations,
)
choices.append(
@ -763,6 +774,42 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
return {"format": {"type": "text"}}
return None
@staticmethod
def _convert_annotations_to_chat_format(
annotations: Optional[List[Any]],
) -> Optional[List[Dict[str, Any]]]:
"""
Convert annotations from Responses API to Chat Completions format.
Annotations are already in compatible format between both APIs,
so we just need to convert Pydantic models to dicts.
"""
if not annotations:
return None
result: List[Dict[str, Any]] = []
for annotation in annotations:
try:
# Convert Pydantic models to dicts (handles both v1 and v2)
if hasattr(annotation, "model_dump"):
annotation_dict = annotation.model_dump()
elif hasattr(annotation, "dict"):
annotation_dict = annotation.dict()
elif isinstance(annotation, dict):
annotation_dict = annotation
else:
# Skip unsupported annotation types
verbose_logger.debug(f"Skipping unsupported annotation type: {type(annotation)}")
continue
result.append(annotation_dict)
except Exception as e:
# Skip malformed annotations
verbose_logger.debug(f"Skipping malformed annotation: {annotation}, error: {e}")
continue
return result if result else None
def _map_responses_status_to_finish_reason(self, status: Optional[str]) -> str:
"""Map responses API status to chat completion finish_reason"""

View file

@ -336,8 +336,8 @@ class SkillsInjectionHook(CustomLogger):
)
# Check if code execution is enabled for this request
litellm_metadata = request_data.get("litellm_metadata", {})
metadata = request_data.get("metadata", {})
litellm_metadata = request_data.get("litellm_metadata") or {}
metadata = request_data.get("metadata") or {}
code_exec_enabled = (
litellm_metadata.get("_litellm_code_execution_enabled") or

View file

@ -1095,3 +1095,186 @@ def test_map_reasoning_effort_adds_summary_detailed():
os.environ["LITELLM_REASONING_AUTO_SUMMARY"] = original_env
elif "LITELLM_REASONING_AUTO_SUMMARY" in os.environ:
del os.environ["LITELLM_REASONING_AUTO_SUMMARY"]
def test_transform_response_preserves_annotations():
"""
Test that annotations from Responses API are preserved when transforming to Chat Completions format.
This is a regression test for the bug where annotations (like url_citation) were being
dropped during the transformation from ResponsesAPIResponse to ModelResponse.
The fix ensures annotations are extracted from ResponseOutputText content items and
passed through to the Message object in the Chat Completions response.
"""
from unittest.mock import Mock
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
from litellm.types.llms.openai import (
InputTokensDetails,
OutputTokensDetails,
ResponseAPIUsage,
ResponsesAPIResponse,
)
from litellm.types.utils import ModelResponse, Usage
handler = LiteLLMResponsesTransformationHandler()
# Create annotations similar to what OpenAI Responses API returns
annotations = [
{
"type": "url_citation",
"start_index": 0,
"end_index": 100,
"title": "Example Article",
"url": "https://example.com/article",
},
{
"type": "url_citation",
"start_index": 101,
"end_index": 200,
"title": "Another Source",
"url": "https://example.com/source",
},
]
# Create output text with annotations
output_text = ResponseOutputText(
annotations=annotations,
text="Here is some information with citations.",
type="output_text",
logprobs=[],
)
# Create output message
output_message = ResponseOutputMessage(
id="msg_test123",
content=[output_text],
role="assistant",
status="completed",
type="message",
)
# Create usage information
usage = ResponseAPIUsage(
input_tokens=10,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
output_tokens=20,
output_tokens_details=OutputTokensDetails(
reasoning_tokens=0, text_tokens=None
),
total_tokens=30,
cost=None,
)
# Create the full ResponsesAPIResponse
raw_response = ResponsesAPIResponse(
id="resp_test123",
created_at=1234567890,
error=None,
incomplete_details=None,
instructions=None,
metadata={},
model="gpt-5.1",
object="response",
output=[output_message],
parallel_tool_calls=True,
temperature=1.0,
tool_choice="auto",
tools=[],
top_p=1.0,
max_output_tokens=None,
previous_response_id=None,
reasoning=None,
status="completed",
text={"format": {"type": "text"}, "verbosity": "medium"},
truncation="disabled",
usage=usage,
user=None,
store=True,
background=False,
billing={"payer": "openai"},
max_tool_calls=None,
prompt_cache_key=None,
safety_identifier=None,
service_tier="default",
top_logprobs=0,
)
# Create empty model_response
model_response = ModelResponse(
id="chatcmpl-test123",
created=1234567890,
model=None,
object="chat.completion",
system_fingerprint=None,
choices=[],
usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0),
)
# Create mock objects for required parameters
logging_obj = Mock()
messages = [{"role": "user", "content": "Tell me about AI"}]
request_data = {"model": "gpt-5.1"}
optional_params = {}
litellm_params = {"acompletion": False, "api_key": None}
encoding = Mock()
# Call transform_response
result = handler.transform_response(
model="gpt-5.1",
raw_response=raw_response,
model_response=model_response,
logging_obj=logging_obj,
request_data=request_data,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
encoding=encoding,
api_key=None,
json_mode=None,
)
# Assertions
assert result.model == "gpt-5.1"
assert len(result.choices) == 1
# Check the choice
choice = result.choices[0]
assert choice.finish_reason == "stop"
assert choice.index == 0
assert choice.message.role == "assistant"
assert choice.message.content == "Here is some information with citations."
# Check that annotations are preserved
assert hasattr(choice.message, "annotations"), "Message should have annotations attribute"
assert choice.message.annotations is not None, "Annotations should not be None"
assert len(choice.message.annotations) == 2, f"Expected 2 annotations, got {len(choice.message.annotations)}"
# Verify annotation content
annotation1 = choice.message.annotations[0]
assert annotation1["type"] == "url_citation"
assert annotation1["title"] == "Example Article"
assert annotation1["url"] == "https://example.com/article"
assert annotation1["start_index"] == 0
assert annotation1["end_index"] == 100
annotation2 = choice.message.annotations[1]
assert annotation2["type"] == "url_citation"
assert annotation2["title"] == "Another Source"
assert annotation2["url"] == "https://example.com/source"
assert annotation2["start_index"] == 101
assert annotation2["end_index"] == 200
# Check usage
assert result.usage.prompt_tokens == 10
assert result.usage.completion_tokens == 20
assert result.usage.total_tokens == 30
print("✓ Annotations from Responses API are correctly preserved in Chat Completions format")