This commit is contained in:
Stephen Chin 2026-09-28 11:30:00 -07:00 • committed by GitHub
commit 4628b6c3fd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 593 additions and 92 deletions

View file

@ -51,7 +51,11 @@ from litellm.types.llms.openai import (
from litellm.types.utils import GenericStreamingChunk, ModelResponseStream
if TYPE_CHECKING:
from openai.types.responses import ResponseInputImageParam, ResponseOutputItem
from openai.types.responses import (
ResponseInputImageParam,
ResponseOutputItem,
ResponseOutputMessage,
)
from openai.types.responses.response_text_config_param import (
ResponseTextConfigParam as ResponseText,
)
@ -658,6 +662,102 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
return request_data
@staticmethod
def _choices_from_response_output_message(
item: "ResponseOutputMessage",
starting_index: int,
reasoning_content: str | None,
pending_reasoning_item: dict[str, Any] | None,
) -> tuple[list[Any], int]:
"""Build one Choices per content block on a ResponseOutputMessage.
The first emitted choice carries the pending reasoning (content + items);
any subsequent content blocks emit choices with reasoning fields cleared,
matching the flush semantics of the original inline loop.
"""
from litellm.types.utils import Choices, Message
new_choices: Final[list[Any]] = []
current_index = starting_index
carry_reasoning_content = reasoning_content
carry_reasoning_item = pending_reasoning_item
for content in item.content:
response_text = getattr(content, "text", "")
raw_annotations = getattr(content, "annotations", None)
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(raw_annotations)
reasoning_items_for_msg = cast(
list[ChatCompletionReasoningItem] | None,
([carry_reasoning_item] if carry_reasoning_item is not None else None),
)
msg = Message(
role=item.role,
content=response_text or "",
reasoning_content=carry_reasoning_content,
annotations=annotations,
reasoning_items=reasoning_items_for_msg,
)
new_choices.append(Choices(message=msg, finish_reason="stop", index=current_index))
carry_reasoning_content = None
carry_reasoning_item = None
current_index += 1
return new_choices, current_index
@staticmethod
def _merge_accumulated_tool_calls_into_choices(
choices: list[Any],
accumulated_tool_calls: list[dict[str, Any]],
reasoning_content: str | None,
pending_reasoning_item: dict[str, Any] | None,
fallback_index: int,
) -> None:
"""Attach accumulated tool_calls to the last text-message choice, or
append a new tool-only choice if no text message was produced.
Backfills reasoning_content and reasoning_items onto the merged choice
so encrypted reasoning survives the assistant+tool_calls merge path.
"""
from litellm.types.utils import Choices, Message
last_msg_choice = next(
(
c
for c in reversed(choices)
if getattr(c, "message", None) is not None and not getattr(c.message, "tool_calls", None)
),
None,
)
if last_msg_choice is None:
reasoning_items_for_msg = cast(
list[ChatCompletionReasoningItem] | None,
([pending_reasoning_item] if pending_reasoning_item is not None else None),
)
msg = Message(
content=None,
tool_calls=accumulated_tool_calls,
reasoning_content=reasoning_content,
reasoning_items=reasoning_items_for_msg,
)
choices.append(Choices(message=msg, finish_reason="tool_calls", index=fallback_index))
return
last_msg_choice.message.tool_calls = accumulated_tool_calls
if getattr(last_msg_choice.message, "content", None) is None:
last_msg_choice.message.content = ""
last_msg_choice.finish_reason = "tool_calls"
needs_reasoning_content_backfill = (
reasoning_content is not None and getattr(last_msg_choice.message, "reasoning_content", None) is None
)
if needs_reasoning_content_backfill:
last_msg_choice.message.reasoning_content = reasoning_content
needs_reasoning_items_backfill = (
pending_reasoning_item is not None and getattr(last_msg_choice.message, "reasoning_items", None) is None
)
if needs_reasoning_items_backfill:
last_msg_choice.message.reasoning_items = cast(
list[ChatCompletionReasoningItem] | None,
[pending_reasoning_item],
)
@staticmethod
def _convert_response_output_to_choices(
output_items: Sequence[object],
@ -686,9 +786,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
except ImportError:
ResponseApplyPatchToolCall = None
from litellm.types.utils import Choices, Message
choices: Final[list[Choices]] = []
choices: Final[list[Any]] = []
index = 0
reasoning_content: str | None = None
pending_reasoning_item: _BuiltReasoningItem | None = None
@ -708,35 +806,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
reasoning_content = " ".join(s["text"] for s in pending_reasoning_item["summary"] if s.get("text"))
elif isinstance(item, ResponseOutputMessage):
for content in item.content:
response_text = getattr(content, "text", "")
# Extract annotations from content if present
raw_annotations = getattr(content, "annotations", None)
annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
raw_annotations
)
msg = Message(
role=item.role,
content=response_text if response_text else "",
reasoning_content=reasoning_content,
annotations=annotations,
reasoning_items=cast(
list[ChatCompletionReasoningItem] | None,
([pending_reasoning_item] if pending_reasoning_item is not None else None),
),
)
choices.append(
Choices(
message=msg,
finish_reason="stop",
index=index,
)
)
reasoning_content = None # flush
pending_reasoning_item = None # flush
index += 1
new_choices, index = LiteLLMResponsesTransformationHandler._choices_from_response_output_message(
item=item,
starting_index=index,
reasoning_content=reasoning_content,
pending_reasoning_item=pending_reasoning_item,
)
choices.extend(new_choices)
reasoning_content = None
pending_reasoning_item = None
elif isinstance(item, ResponseFunctionToolCall):
from litellm.responses.litellm_completion_transformation.transformation import (
@ -784,20 +862,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
else:
pass # don't fail request if item in list is not supported
# If we accumulated tool calls, create a single choice with all of them
if accumulated_tool_calls:
msg = Message(
content=None,
tool_calls=accumulated_tool_calls,
LiteLLMResponsesTransformationHandler._merge_accumulated_tool_calls_into_choices(
choices=choices,
accumulated_tool_calls=accumulated_tool_calls,
reasoning_content=reasoning_content,
reasoning_items=cast(
list[ChatCompletionReasoningItem] | None,
([pending_reasoning_item] if pending_reasoning_item is not None else None),
),
pending_reasoning_item=pending_reasoning_item,
fallback_index=index,
)
choices.append(Choices(message=msg, finish_reason="tool_calls", index=index))
reasoning_content = None
pending_reasoning_item = None
return choices

View file

@ -1,9 +1,7 @@
import datetime
import json
import os
import unittest
from typing import TYPE_CHECKING, Final, List, Literal, Optional, Tuple, get_args
from unittest.mock import ANY, MagicMock, Mock, patch
from typing import TYPE_CHECKING, Final, Literal, get_args
from unittest.mock import MagicMock, Mock, patch
import httpx
import pytest
@ -55,7 +53,6 @@ def test_convert_chat_completion_messages_to_responses_api_image_input():
assert user_content in response_str
assert user_image in response_str
print("response: ", response)
assert response[0]["content"][1]["image_url"] == user_image
@ -146,8 +143,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag
), "image_url should be a flat string, not a nested object"
assert "detail" in image_item, "detail field should be present"
print("✓ Tool result with image correctly transformed to Responses API format")
def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text():
"""
@ -224,10 +219,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text
text_item["text"] == "15 degrees"
), f"Expected text '15 degrees', got '{text_item.get('text')}'"
print(
"✓ Tool result with text correctly transformed to use input_text for Responses API format"
)
def test_openai_responses_chunk_parser_reasoning_summary():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
@ -512,8 +503,6 @@ and I learn to carry this small calm home."""
# Check reasoning content
assert choice.message.reasoning_content == reasoning_summary.text
print("✓ transform_response correctly handled reasoning items and output messages")
def _make_empty_responses_api_response(model: str = "gpt-5.4"):
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
@ -983,10 +972,6 @@ def test_transform_request_single_char_keys_not_matched():
assert result_correct.get("metadata") == {"user_id": "123"}
assert result_correct.get("previous_response_id") == "resp_abc"
print(
"✓ Single-character keys are not incorrectly matched to metadata/previous_response_id"
)
# =============================================================================
# Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls
@ -1390,8 +1375,6 @@ def test_tool_message_output_uses_input_text_not_output_text():
), f"Expected input_text, got {output[0].get('type')}"
assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}'
print("✓ Tool message output correctly uses input_text type")
def test_multiple_tool_calls_in_single_choice():
"""
@ -1533,8 +1516,6 @@ def test_multiple_tool_calls_in_single_choice():
assert tool_calls[2]["id"] == "call_horoscope"
assert tool_calls[2]["function"]["name"] == "get_horoscope"
print("✓ Multiple tool calls are correctly grouped in a single choice")
def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
"""
@ -1547,7 +1528,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
When flag is enabled (flag=True or env var), summary="detailed" is added.
"""
import litellm
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
@ -1576,9 +1556,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
"summary" not in result
), f"Summary should NOT be present by default for effort={effort}"
print(
f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)"
)
# Test 2: With flag enabled - summary IS added
litellm.reasoning_auto_summary = True
@ -1592,9 +1569,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
result["summary"] == "detailed"
), f"Summary should be 'detailed' when flag is enabled for effort={effort}"
print(
f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)"
)
# Test 3: With env var enabled (flag disabled) - summary IS added
litellm.reasoning_auto_summary = False
@ -1604,7 +1578,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
assert (
result["summary"] == "detailed"
), "Summary should be 'detailed' when env var is enabled"
print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly")
# Test 4: Dict input is passed through as-is (no modification)
litellm.reasoning_auto_summary = False
@ -1615,11 +1588,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
result_dict = handler._map_reasoning_effort(dict_input)
assert result_dict["effort"] == "high"
assert result_dict["summary"] == "custom_summary"
print("✓ Dict input is passed through without modification")
print(
"✓ All reasoning_effort behaviors work correctly with flag/env var control"
)
finally:
# Restore original values
@ -1812,10 +1780,6 @@ def test_transform_response_preserves_annotations():
assert result.usage.completion_tokens == 20
assert result.usage.total_tokens == 30
print(
"✓ Annotations from Responses API are correctly preserved in Chat Completions format"
)
def test_apply_patch_tool_call_converted_to_chat_completion_tool_call():
"""
@ -2109,10 +2073,6 @@ def test_multi_tool_call_stream_no_premature_finish():
f"— only response.completed should terminate the stream"
)
print(
"✓ Multi-tool-call stream completes without premature finish_reason termination"
)
# =============================================================================
# Tests for issue #21331: Parallel tool call indices in streaming
@ -2425,10 +2385,6 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
1,
}, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}"
print(
"✓ Parallel tool calls with split argument deltas stream correctly end-to-end"
)
def test_map_optional_params_preserves_reasoning_summary():
"""Test that reasoning_effort dict with summary field is preserved.
@ -2911,6 +2867,165 @@ def test_reasoning_items_streaming_emitted_on_response_completed():
assert ri["summary"][0]["text"] == summary_text
def test_reasoning_items_preserved_when_merged_with_tool_calls():
"""
Regression: when a Responses turn contains [message, reasoning, function_call]
(reasoning item arriving AFTER the assistant message), the merge path that
attaches accumulated tool_calls onto the last message choice must also backfill
the structured ``reasoning_items`` (with ``encrypted_content``) onto that
message; otherwise the encrypted reasoning payload needed to round-trip on the
next turn is silently dropped. The flattened ``reasoning_content`` string is
already backfilled by the existing code; ``reasoning_items`` currently is not.
"""
from unittest.mock import Mock
from openai.types.responses import (
ResponseFunctionToolCall,
ResponseOutputMessage,
ResponseOutputText,
)
from openai.types.responses.response_reasoning_item import (
ResponseReasoningItem,
Summary,
)
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
from litellm.types.llms.openai import (
InputTokensDetails,
OutputTokensDetails,
ResponseAPIUsage,
ResponsesAPIResponse,
)
from litellm.types.utils import ModelResponse, Usage
handler = LiteLLMResponsesTransformationHandler()
encrypted = "gAAAAABpw5xyz789FAKE=="
summary_text = "deciding which city"
preamble = "Let me check the weather."
output_message = ResponseOutputMessage(
id="msg_merge001",
content=[
ResponseOutputText(
annotations=[],
text=preamble,
type="output_text",
logprobs=[],
)
],
role="assistant",
status="completed",
type="message",
)
reasoning_item = ResponseReasoningItem(
id="rs_merge001",
summary=[Summary(text=summary_text, type="summary_text")],
type="reasoning",
content=None,
encrypted_content=encrypted,
status=None,
)
function_call_item = ResponseFunctionToolCall(
id="fc_merge001",
type="function_call",
status="completed",
arguments='{"city": "Paris"}',
call_id="call_paris_merge",
name="get_weather",
)
usage = ResponseAPIUsage(
input_tokens=10,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
output_tokens=20,
output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None),
total_tokens=30,
cost=None,
)
raw_response = ResponsesAPIResponse(
id="resp_merge001",
created_at=1234567890,
error=None,
incomplete_details=None,
instructions=None,
metadata={},
model="gpt-5-mini",
object="response",
output=[output_message, reasoning_item, function_call_item],
parallel_tool_calls=True,
temperature=1.0,
tool_choice="auto",
tools=[],
top_p=1.0,
max_output_tokens=None,
previous_response_id=None,
reasoning={"effort": "low", "summary": "detailed"},
status="completed",
text={"format": {"type": "text"}, "verbosity": "medium"},
truncation="disabled",
usage=usage,
user=None,
store=True,
background=False,
billing={"payer": "developer"},
max_tool_calls=None,
prompt_cache_key=None,
safety_identifier=None,
service_tier="default",
top_logprobs=0,
)
model_response = ModelResponse(
id="chatcmpl-merge001",
created=1234567890,
model=None,
object="chat.completion",
system_fingerprint=None,
choices=[],
usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0),
)
result = handler.transform_response(
model="gpt-5-mini",
raw_response=raw_response,
model_response=model_response,
logging_obj=Mock(),
request_data={"model": "gpt-5-mini"},
messages=[{"role": "user", "content": "What's the weather in Paris?"}],
optional_params={},
litellm_params={},
encoding=Mock(),
)
assert len(result.choices) == 1, (
f"merge must collapse into a single choice, got {len(result.choices)}"
)
choice = result.choices[0]
msg = choice.message
assert msg.tool_calls is not None and len(msg.tool_calls) == 1
assert msg.tool_calls[0]["function"]["name"] == "get_weather"
assert msg.tool_calls[0]["function"]["arguments"] == '{"city": "Paris"}'
assert choice.finish_reason == "tool_calls"
assert msg.content == preamble, "assistant preamble text must be preserved"
assert msg.reasoning_content == summary_text, (
"reasoning_content must be backfilled onto the merged choice"
)
assert msg.reasoning_items is not None, (
"reasoning_items must be backfilled onto the merged choice so encrypted_content "
"can round-trip to the provider on the next turn"
)
assert len(msg.reasoning_items) == 1
assert msg.reasoning_items[0]["encrypted_content"] == encrypted
def test_streaming_function_call_tool_id_for_degenerate_call_id():
"""In streaming, Bedrock Mantle's function_call event carries a unique ``id``
(``fc_...``) and a non-unique, index-based ``call_id`` (``call_0``). For that
@ -2945,6 +3060,320 @@ def test_streaming_function_call_tool_id_for_degenerate_call_id():
assert stream_tool_id("fc_2", "call_tokyo") == "call_tokyo"
def _build_message_plus_tool_call_response(
output_items,
model="gpt-4o",
):
"""Helper that wraps output_items into a minimal ResponsesAPIResponse and returns the
transformed ModelResponse produced by LiteLLMResponsesTransformationHandler."""
from unittest.mock import Mock
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
from litellm.types.llms.openai import (
InputTokensDetails,
OutputTokensDetails,
ResponseAPIUsage,
ResponsesAPIResponse,
)
from litellm.types.utils import ModelResponse, Usage
usage = ResponseAPIUsage(
input_tokens=10,
input_tokens_details=InputTokensDetails(cached_tokens=0),
output_tokens=20,
output_tokens_details=OutputTokensDetails(reasoning_tokens=0),
total_tokens=30,
)
raw_response = ResponsesAPIResponse(
id="resp_bug18401",
created_at=1234567890,
error=None,
incomplete_details=None,
instructions=None,
metadata={},
model=model,
object="response",
output=output_items,
parallel_tool_calls=True,
temperature=1.0,
tool_choice="auto",
tools=[],
top_p=1.0,
max_output_tokens=None,
previous_response_id=None,
reasoning=None,
status="completed",
text=None,
truncation="disabled",
usage=usage,
user=None,
store=True,
background=False,
)
model_response = ModelResponse(
id="chatcmpl-bug18401",
created=1234567890,
model=None,
object="chat.completion",
choices=[],
usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0),
)
handler = LiteLLMResponsesTransformationHandler()
return handler.transform_response(
model=model,
raw_response=raw_response,
model_response=model_response,
logging_obj=Mock(),
request_data={"model": model},
messages=[{"role": "user", "content": "test"}],
optional_params={},
litellm_params={},
encoding=Mock(),
)
def _make_output_message(text, message_id="msg_bug18401"):
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
return ResponseOutputMessage(
id=message_id,
content=[
ResponseOutputText(
annotations=[],
text=text,
type="output_text",
logprobs=[],
)
],
role="assistant",
status="completed",
type="message",
)
def _make_function_tool_call(call_id, name, arguments, tc_id="fc_bug18401"):
from openai.types.responses import ResponseFunctionToolCall
return ResponseFunctionToolCall(
id=tc_id,
type="function_call",
status="completed",
arguments=arguments,
call_id=call_id,
name=name,
)
def test_message_plus_function_call_merged_into_single_choice():
"""Regression: a Responses turn that contains both a message and a function_call must
collapse into a single Chat Completions choice, because Chat Completions clients only
read choices[0]. Prior to the fix the bridge emitted two choices (content in [0],
tool_calls in [1]) which caused tools to be silently ignored downstream."""
result = _build_message_plus_tool_call_response(
output_items=[
_make_output_message("Fetching the weather now."),
_make_function_tool_call(
call_id="call_paris",
name="get_weather",
arguments='{"location": "Paris"}',
),
]
)
assert len(result.choices) == 1, f"Expected 1 choice, got {len(result.choices)}"
choice = result.choices[0]
assert choice.finish_reason == "tool_calls"
assert choice.message.content == "Fetching the weather now."
tool_calls = choice.message.tool_calls
assert tool_calls is not None and len(tool_calls) == 1
assert tool_calls[0]["id"] == "call_paris"
assert tool_calls[0]["function"]["name"] == "get_weather"
assert tool_calls[0]["function"]["arguments"] == '{"location": "Paris"}'
def test_tool_only_turn_unchanged():
"""Guard against regressing the pre-fix behaviour for tool-only turns: when the
Responses output has no assistant message, a fresh Choice must still be appended
with the accumulated tool_calls and finish_reason='tool_calls'."""
result = _build_message_plus_tool_call_response(
output_items=[
_make_function_tool_call(
call_id="call_only",
name="get_weather",
arguments='{"location": "Tokyo"}',
),
]
)
assert len(result.choices) == 1
choice = result.choices[0]
assert choice.finish_reason == "tool_calls"
tool_calls = choice.message.tool_calls
assert tool_calls is not None and len(tool_calls) == 1
assert tool_calls[0]["id"] == "call_only"
def test_message_only_turn_unchanged():
"""Guard: a message-only Responses turn must still produce a single choice with
finish_reason='stop' after the merge fix (no accumulated_tool_calls path taken)."""
result = _build_message_plus_tool_call_response(
output_items=[_make_output_message("Just a plain answer.")]
)
assert len(result.choices) == 1
choice = result.choices[0]
assert choice.finish_reason == "stop"
assert choice.message.content == "Just a plain answer."
assert not choice.message.tool_calls
def test_multiple_function_calls_after_message_merged():
"""When a message is followed by multiple parallel function_calls, all of them must
land on the single message-carrying choice (Chat Completions groups them all under
one message)."""
result = _build_message_plus_tool_call_response(
output_items=[
_make_output_message("Fetching two cities in parallel."),
_make_function_tool_call(
call_id="call_paris",
name="get_weather",
arguments='{"location": "Paris"}',
tc_id="fc_1",
),
_make_function_tool_call(
call_id="call_tokyo",
name="get_weather",
arguments='{"location": "Tokyo"}',
tc_id="fc_2",
),
]
)
assert len(result.choices) == 1
choice = result.choices[0]
assert choice.finish_reason == "tool_calls"
assert choice.message.content == "Fetching two cities in parallel."
tool_calls = choice.message.tool_calls
assert tool_calls is not None and len(tool_calls) == 2
assert {tc["id"] for tc in tool_calls} == {"call_paris", "call_tokyo"}
assert all(tc["function"]["name"] == "get_weather" for tc in tool_calls)
def test_multiple_message_items_tool_calls_collapse_onto_last():
"""When the Responses output contains two assistant messages before a function_call,
the accumulated tool_calls must attach to the LAST (most recent) message choice, not
the first, because that is the one the model was on when it decided to call the tool.
The earlier message choice must remain a plain text turn with finish_reason='stop'."""
result = _build_message_plus_tool_call_response(
output_items=[
_make_output_message("first", message_id="msg_first"),
_make_output_message("second", message_id="msg_second"),
_make_function_tool_call(
call_id="call_after_second",
name="get_weather",
arguments='{"location": "Paris"}',
),
]
)
assert len(result.choices) == 2
first_choice = next(c for c in result.choices if c.message.content == "first")
second_choice = next(c for c in result.choices if c.message.content == "second")
assert first_choice.finish_reason == "stop"
assert not first_choice.message.tool_calls
assert second_choice.finish_reason == "tool_calls"
tool_calls = second_choice.message.tool_calls
assert tool_calls is not None and len(tool_calls) == 1
assert tool_calls[0]["id"] == "call_after_second"
def test_streaming_text_plus_function_call_lands_on_choice_index_zero():
"""Streaming counterpart to the merged-choice fix: when a message and a function_call
arrive in the same Responses turn, both the content deltas and the tool_call deltas
must be emitted on choice index 0, and the terminal response.completed chunk must
carry finish_reason='tool_calls'. This mirrors the non-streaming merge behaviour so
that Chat Completions clients that only read choices[0] see both the text and the
tool call."""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
chunks = [
{"type": "response.output_text.delta", "delta": "Fetching "},
{"type": "response.output_text.delta", "delta": "the weather."},
{
"type": "response.output_item.done",
"item": {"type": "message", "content": []},
},
{
"type": "response.output_item.added",
"item": {
"type": "function_call",
"name": "get_weather",
"call_id": "call_paris",
"id": "fc_1",
},
},
{
"type": "response.function_call_arguments.delta",
"delta": '{"location":"Paris"}',
},
{
"type": "response.output_item.done",
"item": {
"type": "function_call",
"name": "get_weather",
"call_id": "call_paris",
"arguments": '{"location":"Paris"}',
},
},
{
"type": "response.completed",
"response": {
"output": [
{"type": "message"},
{"type": "function_call", "call_id": "call_paris"},
]
},
},
]
results = [iterator.chunk_parser(chunk) for chunk in chunks]
for idx, r in enumerate(results):
assert len(r.choices) == 1, f"chunk {idx} produced {len(r.choices)} choices"
assert r.choices[0].index == 0, (
f"chunk {idx} landed on index {r.choices[0].index}, expected 0"
)
text_deltas = [r.choices[0].delta.content for r in results[:2]]
assert text_deltas == ["Fetching ", "the weather."]
tool_added = results[3].choices[0].delta.tool_calls
assert tool_added and tool_added[0]["function"]["name"] == "get_weather"
tool_arg_delta = results[4].choices[0].delta.tool_calls
assert (
tool_arg_delta
and tool_arg_delta[0]["function"]["arguments"] == '{"location":"Paris"}'
)
terminal = results[-1]
assert terminal.choices[0].finish_reason == "tool_calls"
def test_streaming_chunks_share_one_chat_completion_id():
"""Every chunk of one streamed chat completion must carry the same ``id``, per the
OpenAI spec. The bridge builds a fresh ``ModelResponseStream`` per Responses event,
@ -3567,8 +3996,8 @@ async def test_acompletion_bridge_normalizes_tool_choice_on_the_wire(
def _make_incomplete_responses_api_response(
incomplete_reason: Optional[str],
output: "List[ResponseOutputItem]",
incomplete_reason: str | None,
output: "list[ResponseOutputItem]",
status: Literal["completed", "incomplete"] = "incomplete",
empty_incomplete_details: bool = False,
) -> "ResponsesAPIResponse":