This commit is contained in:
JingHao-Leon 2026-10-04 12:47:45 -07:00 • committed by GitHub
commit dd7c8b9861
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 228 additions and 285 deletions

View file

@ -1591,6 +1591,35 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
)
else:
raise ValueError(f"Chat provider: Invalid text delta {parsed_chunk}")
elif event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED:
# A url_citation / file_citation arrived for the in-flight message.
# Emit it on the delta exactly once in Chat Completions format —
# this is the only event that carries it, so later
# output_item.done / response.completed events must not repeat it
# or accumulating clients would see duplicates (issue #43817).
raw_annotation: Final = parsed_chunk.get("annotation", None)
annotation: Final = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(
[raw_annotation] if raw_annotation is not None else None
)
if annotation:
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(annotations=annotation),
finish_reason=None,
)
]
)
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(),
finish_reason=None,
)
]
)
elif event_type == "response.reasoning_summary_text.delta":
content_part = parsed_chunk.get("delta", None)
if content_part:

View file

@ -135,9 +135,7 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag
function_call_output = item
break
assert (
function_call_output is not None
), "function_call_output not found in response"
assert function_call_output is not None, "function_call_output not found in response"
assert function_call_output["call_id"] == "call_abc123"
# Check that the output is correctly transformed
@ -147,12 +145,8 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag
image_item = output[0]
# Should be transformed to Responses API format
assert (
image_item["type"] == "input_image"
), f"Expected type 'input_image', got '{image_item.get('type')}'"
assert (
image_item["image_url"] == test_image_base64
), "image_url should be a flat string, not a nested object"
assert image_item["type"] == "input_image", f"Expected type 'input_image', got '{image_item.get('type')}'"
assert image_item["image_url"] == test_image_base64, "image_url should be a flat string, not a nested object"
assert "detail" in image_item, "detail field should be present"
print("✓ Tool result with image correctly transformed to Responses API format")
@ -214,9 +208,7 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text
function_call_output = item
break
assert (
function_call_output is not None
), "function_call_output not found in response"
assert function_call_output is not None, "function_call_output not found in response"
assert function_call_output["call_id"] == "call_abc123"
# Check that the output is correctly transformed to use input_text, not output_text
@ -226,16 +218,12 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text
text_item = output[0]
# Should be transformed to use input_text for tool results in Responses API format
assert (
text_item["type"] == "input_text"
), f"Expected type 'input_text' for tool result, got '{text_item.get('type')}'"
assert (
text_item["text"] == "15 degrees"
), f"Expected text '15 degrees', got '{text_item.get('text')}'"
print(
"✓ Tool result with text correctly transformed to use input_text for Responses API format"
assert text_item["type"] == "input_text", (
f"Expected type 'input_text' for tool result, got '{text_item.get('type')}'"
)
assert text_item["text"] == "15 degrees", f"Expected text '15 degrees', got '{text_item.get('text')}'"
print("✓ Tool result with text correctly transformed to use input_text for Responses API format")
def test_openai_responses_chunk_parser_reasoning_summary():
@ -244,9 +232,7 @@ def test_openai_responses_chunk_parser_reasoning_summary():
)
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"delta": "**Compar",
@ -278,9 +264,7 @@ def test_chunk_parser_string_output_text_delta_produces_text():
)
from litellm.types.utils import ModelResponseStream
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {"type": "response.output_text.delta", "delta": "literal text"}
@ -301,9 +285,7 @@ def test_chunk_parser_enum_output_text_delta_produces_text():
from litellm.types.llms.openai import ResponsesAPIStreamEvents
from litellm.types.utils import ModelResponseStream
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {"type": ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA, "delta": "enum text"}
@ -324,9 +306,7 @@ def test_chunk_parser_function_call_added_produces_tool_use():
from litellm.types.llms.openai import ResponsesAPIStreamEvents
from litellm.types.utils import ModelResponseStream
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED,
@ -411,9 +391,7 @@ Tomorrow will bring its petitions and promises,
but for now the city breathes slow and wide,
and I learn to carry this small calm home."""
output_text = ResponseOutputText(
annotations=[], text=poem_text, type="output_text", logprobs=[]
)
output_text = ResponseOutputText(annotations=[], text=poem_text, type="output_text", logprobs=[])
output_message = ResponseOutputMessage(
id="msg_04c8021b8b3188a00068e9ae0b92f4819dac64d85b4abb67ec",
content=[output_text],
@ -425,9 +403,7 @@ and I learn to carry this small calm home."""
# Create usage information
usage = ResponseAPIUsage(
input_tokens=16,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None),
output_tokens=195,
output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None),
total_tokens=211,
@ -776,11 +752,7 @@ def test_recover_output_items_merges_text_only_items_at_distinct_indices():
]
)
recovered = (
LiteLLMResponsesTransformationHandler._recover_output_items_from_raw_sse(
raw_sse
)
)
recovered = LiteLLMResponsesTransformationHandler._recover_output_items_from_raw_sse(raw_sse)
assert len(recovered) == 2
assert recovered[0]["id"] == "msg_item_0"
@ -918,9 +890,7 @@ def test_transform_request_system_only_message_maps_to_system_input_item():
{
"type": "message",
"role": "system",
"content": [
{"type": "input_text", "text": "You are a helpful assistant."}
],
"content": [{"type": "input_text", "text": "You are a helpful assistant."}],
}
]
# System content lives in input only; not duplicated into instructions.
@ -992,9 +962,7 @@ def test_transform_request_single_char_keys_not_matched():
assert result_correct.get("metadata") == {"user_id": "123"}
assert result_correct.get("previous_response_id") == "resp_abc"
print(
"✓ Single-character keys are not incorrectly matched to metadata/previous_response_id"
)
print("✓ Single-character keys are not incorrectly matched to metadata/previous_response_id")
# =============================================================================
@ -1014,9 +982,7 @@ def test_message_done_does_not_emit_is_finished():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.output_item.done",
@ -1028,9 +994,9 @@ def test_message_done_does_not_emit_is_finished():
# After the fix, message completion should NOT set finish_reason
# ModelResponseStream doesn't have is_finished - check finish_reason instead
assert len(result.choices) > 0, "result should have choices"
assert (
result.choices[0].finish_reason is None or result.choices[0].finish_reason == ""
), "message completion should not emit finish_reason"
assert result.choices[0].finish_reason is None or result.choices[0].finish_reason == "", (
"message completion should not emit finish_reason"
)
def test_response_completed_emits_is_finished():
@ -1042,9 +1008,7 @@ def test_response_completed_emits_is_finished():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {"type": "response.completed"}
@ -1052,9 +1016,7 @@ def test_response_completed_emits_is_finished():
# response.completed should emit finish_reason='stop'
assert len(result.choices) > 0, "result should have choices"
assert (
result.choices[0].finish_reason == "stop"
), "response.completed should emit finish_reason='stop'"
assert result.choices[0].finish_reason == "stop", "response.completed should emit finish_reason='stop'"
def test_response_completed_with_function_calls_emits_tool_calls_finish_reason():
@ -1073,9 +1035,7 @@ def test_response_completed_with_function_calls_emits_tool_calls_finish_reason()
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
# Simulate a response.completed event with function_call in output
# This matches what Azure/OpenAI sends for gpt-5.1-codex-mini and similar models
@ -1101,9 +1061,9 @@ def test_response_completed_with_function_calls_emits_tool_calls_finish_reason()
# response.completed with function_call should emit finish_reason='tool_calls'
assert len(result.choices) > 0, "result should have choices"
assert (
result.choices[0].finish_reason == "tool_calls"
), "response.completed with function_call output should emit finish_reason='tool_calls'"
assert result.choices[0].finish_reason == "tool_calls", (
"response.completed with function_call output should emit finish_reason='tool_calls'"
)
def test_response_completed_with_message_only_emits_stop_finish_reason():
@ -1114,9 +1074,7 @@ def test_response_completed_with_message_only_emits_stop_finish_reason():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
# Simulate a response.completed event with only message output
chunk = {
@ -1140,9 +1098,9 @@ def test_response_completed_with_message_only_emits_stop_finish_reason():
# response.completed with only message should emit finish_reason='stop'
assert len(result.choices) > 0, "result should have choices"
assert (
result.choices[0].finish_reason == "stop"
), "response.completed with only message output should emit finish_reason='stop'"
assert result.choices[0].finish_reason == "stop", (
"response.completed with only message output should emit finish_reason='stop'"
)
def test_response_completed_preserves_usage_with_cached_tokens():
@ -1158,9 +1116,7 @@ def test_response_completed_preserves_usage_with_cached_tokens():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.completed",
@ -1189,18 +1145,12 @@ def test_response_completed_preserves_usage_with_cached_tokens():
result = iterator.chunk_parser(chunk)
assert result.usage is not None, "usage should be set on response.completed chunk"
assert (
result.usage.prompt_tokens == 1226
), "prompt_tokens should map from input_tokens"
assert (
result.usage.completion_tokens == 5
), "completion_tokens should map from output_tokens"
assert (
result.usage.prompt_tokens_details is not None
), "prompt_tokens_details should be set"
assert (
result.usage.prompt_tokens_details.cached_tokens == 1024
), "cached_tokens should be preserved from input_tokens_details"
assert result.usage.prompt_tokens == 1226, "prompt_tokens should map from input_tokens"
assert result.usage.completion_tokens == 5, "completion_tokens should map from output_tokens"
assert result.usage.prompt_tokens_details is not None, "prompt_tokens_details should be set"
assert result.usage.prompt_tokens_details.cached_tokens == 1024, (
"cached_tokens should be preserved from input_tokens_details"
)
def test_function_call_done_emits_is_finished():
@ -1214,9 +1164,7 @@ def test_function_call_done_emits_is_finished():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.output_item.done",
@ -1236,9 +1184,9 @@ def test_function_call_done_emits_is_finished():
"output_item.done for function_call must not emit finish_reason; "
"response.completed is responsible for the terminal finish_reason"
)
assert not result.choices[
0
].delta.tool_calls, "output_item.done for function_call must not include a duplicate tool_calls delta"
assert not result.choices[0].delta.tool_calls, (
"output_item.done for function_call must not include a duplicate tool_calls delta"
)
def test_text_plus_tool_calls_sequence():
@ -1253,9 +1201,7 @@ def test_text_plus_tool_calls_sequence():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
# Simulate the sequence from OpenAI Responses API
chunks = [
@ -1294,28 +1240,23 @@ def test_text_plus_tool_calls_sequence():
# Check message done (index 2) does NOT have finish_reason set
message_done_result = results[2]
assert len(message_done_result.choices) > 0, "message done should have choices"
assert (
message_done_result.choices[0].finish_reason is None
or message_done_result.choices[0].finish_reason == ""
), "message done should not have finish_reason"
assert message_done_result.choices[0].finish_reason is None or message_done_result.choices[0].finish_reason == "", (
"message done should not have finish_reason"
)
# Check function_call done (index 5) does NOT have finish_reason set
# (response.completed is responsible for the terminal finish_reason)
function_done_result = results[5]
assert (
len(function_done_result.choices) > 0
), "function_call done should have choices"
assert (
function_done_result.choices[0].finish_reason is None
), "output_item.done for function_call must not emit finish_reason"
assert len(function_done_result.choices) > 0, "function_call done should have choices"
assert function_done_result.choices[0].finish_reason is None, (
"output_item.done for function_call must not emit finish_reason"
)
# Check response.completed (index 6) has finish_reason='stop'
# (the mock chunk has no nested 'response' data, so has_function_calls is False → 'stop')
completed_result = results[6]
assert len(completed_result.choices) > 0, "response.completed should have choices"
assert (
completed_result.choices[0].finish_reason == "stop"
), "response.completed should have finish_reason='stop'"
assert completed_result.choices[0].finish_reason == "stop", "response.completed should have finish_reason='stop'"
# =============================================================================
@ -1332,7 +1273,11 @@ def test_developer_message_content_uses_input_text():
assert instructions is None
assert input_items == [
{"type": "message", "role": "developer", "content": [{"type": "input_text", "text": "Always answer in French."}]}
{
"type": "message",
"role": "developer",
"content": [{"type": "input_text", "text": "Always answer in French."}],
}
]
@ -1394,9 +1339,7 @@ def test_tool_message_output_uses_input_text_not_output_text():
output = function_call_output["output"]
assert isinstance(output, list), f"output should be a list, got {type(output)}"
assert len(output) == 1
assert (
output[0]["type"] == "input_text"
), f"Expected input_text, got {output[0].get('type')}"
assert output[0]["type"] == "input_text", f"Expected input_text, got {output[0].get('type')}"
assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}'
print("✓ Tool message output correctly uses input_text type")
@ -1581,13 +1524,9 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
assert result is not None, f"Result should not be None for effort={effort}"
assert result["effort"] == effort, f"Effort should be {effort}"
assert (
"summary" not in result
), f"Summary should NOT be present by default for effort={effort}"
assert "summary" not in result, f"Summary should NOT be present by default for effort={effort}"
print(
f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)"
)
print(f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)")
# Test 2: With flag enabled - summary IS added
litellm.reasoning_auto_summary = True
@ -1597,9 +1536,9 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
assert result is not None, f"Result should not be None for effort={effort}"
assert result["effort"] == effort, f"Effort should be {effort}"
assert (
result["summary"] == "detailed"
), f"Summary should be 'detailed' when flag is enabled for effort={effort}"
assert result["summary"] == "detailed", (
f"Summary should be 'detailed' when flag is enabled for effort={effort}"
)
print(
f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)"
@ -1610,9 +1549,7 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
monkeypatch.setenv("LITELLM_REASONING_AUTO_SUMMARY", "true")
result = handler._map_reasoning_effort("high")
assert (
result["summary"] == "detailed"
), "Summary should be 'detailed' when env var is enabled"
assert result["summary"] == "detailed", "Summary should be 'detailed' when env var is enabled"
print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly")
# Test 4: Dict input is passed through as-is (no modification)
@ -1626,9 +1563,7 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch):
assert result_dict["summary"] == "custom_summary"
print("✓ Dict input is passed through without modification")
print(
"✓ All reasoning_effort behaviors work correctly with flag/env var control"
)
print("✓ All reasoning_effort behaviors work correctly with flag/env var control")
finally:
# Restore original values
@ -1704,9 +1639,7 @@ def test_transform_response_preserves_annotations():
# Create usage information
usage = ResponseAPIUsage(
input_tokens=10,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None),
output_tokens=20,
output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None),
total_tokens=30,
@ -1793,13 +1726,9 @@ def test_transform_response_preserves_annotations():
assert choice.message.content == "Here is some information with citations."
# Check that annotations are preserved
assert hasattr(
choice.message, "annotations"
), "Message should have annotations attribute"
assert hasattr(choice.message, "annotations"), "Message should have annotations attribute"
assert choice.message.annotations is not None, "Annotations should not be None"
assert (
len(choice.message.annotations) == 2
), f"Expected 2 annotations, got {len(choice.message.annotations)}"
assert len(choice.message.annotations) == 2, f"Expected 2 annotations, got {len(choice.message.annotations)}"
# Verify annotation content
annotation1 = choice.message.annotations[0]
@ -1821,9 +1750,7 @@ def test_transform_response_preserves_annotations():
assert result.usage.completion_tokens == 20
assert result.usage.total_tokens == 30
print(
"✓ Annotations from Responses API are correctly preserved in Chat Completions format"
)
print("✓ Annotations from Responses API are correctly preserved in Chat Completions format")
def test_apply_patch_tool_call_converted_to_chat_completion_tool_call():
@ -1988,9 +1915,7 @@ def test_multi_tool_call_stream_no_premature_finish():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunks = [
# 0: response created
@ -2066,12 +1991,10 @@ def test_multi_tool_call_stream_no_premature_finish():
r = results[done_idx]
assert r is not None, f"{label}: chunk_parser must return a result"
assert len(r.choices) > 0, f"{label}: result must have choices"
assert (
r.choices[0].finish_reason is None
), f"{label}: output_item.done must not emit finish_reason (stream would terminate prematurely)"
assert not r.choices[
0
].delta.tool_calls, (
assert r.choices[0].finish_reason is None, (
f"{label}: output_item.done must not emit finish_reason (stream would terminate prematurely)"
)
assert not r.choices[0].delta.tool_calls, (
f"{label}: output_item.done must not include a duplicate tool_calls delta"
)
@ -2083,12 +2006,8 @@ def test_multi_tool_call_stream_no_premature_finish():
r = results[added_idx]
if r is not None and r.choices and r.choices[0].delta.tool_calls:
tc = r.choices[0].delta.tool_calls[0]
assert (
tc.function.name == expected_name
), f"output_item.added for {expected_name}: tool_call name mismatch"
assert (
tc.id == expected_call_id
), f"output_item.added for {expected_name}: call_id mismatch"
assert tc.function.name == expected_name, f"output_item.added for {expected_name}: tool_call name mismatch"
assert tc.id == expected_call_id, f"output_item.added for {expected_name}: call_id mismatch"
# 3. argument delta events (indices 2 and 5) should carry arguments
for delta_idx, expected_args, label in [
@ -2098,17 +2017,15 @@ def test_multi_tool_call_stream_no_premature_finish():
r = results[delta_idx]
if r is not None and r.choices and r.choices[0].delta.tool_calls:
tc = r.choices[0].delta.tool_calls[0]
assert (
tc.function.arguments == expected_args
), f"{label}: argument delta mismatch"
assert tc.function.arguments == expected_args, f"{label}: argument delta mismatch"
# 4. Only response.completed (index 7) emits the terminal finish_reason
completed_result = results[7]
assert completed_result is not None, "response.completed must return a result"
assert len(completed_result.choices) > 0, "response.completed must have choices"
assert (
completed_result.choices[0].finish_reason == "tool_calls"
), "response.completed with function_call outputs must emit finish_reason='tool_calls'"
assert completed_result.choices[0].finish_reason == "tool_calls", (
"response.completed with function_call outputs must emit finish_reason='tool_calls'"
)
# 5. No chunk before the last one should have finish_reason set
for idx, r in enumerate(results[:-1]):
@ -2118,9 +2035,7 @@ def test_multi_tool_call_stream_no_premature_finish():
f"— only response.completed should terminate the stream"
)
print(
"✓ Multi-tool-call stream completes without premature finish_reason termination"
)
print("✓ Multi-tool-call stream completes without premature finish_reason termination")
# =============================================================================
@ -2201,16 +2116,13 @@ def test_streaming_parallel_tool_calls_have_distinct_indices():
]
for chunk in chunks:
result = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
chunk
)
result = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk)
expected_index = chunk["output_index"]
for choice in result.choices:
if choice.delta.tool_calls:
for tc in choice.delta.tool_calls:
assert tc.index == expected_index, (
f"Event {chunk['type']}: expected tool_call.index={expected_index}, "
f"got {tc.index}"
f"Event {chunk['type']}: expected tool_call.index={expected_index}, got {tc.index}"
)
@ -2338,9 +2250,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
},
]
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
results = [iterator.chunk_parser(chunk) for chunk in chunks]
# 1. output_item.done events (indices 4 and 8) must NOT emit finish_reason
@ -2352,9 +2262,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
f"{label}: output_item.done must not emit finish_reason "
f"(would prematurely terminate stream before subsequent tool calls arrive)"
)
assert not r.choices[
0
].delta.tool_calls, (
assert not r.choices[0].delta.tool_calls, (
f"{label}: output_item.done must not emit a duplicate tool_calls delta"
)
@ -2388,19 +2296,15 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
for tc in tool_calls:
if tc.function and tc.function.arguments:
idx = tc.index
assembled_args[idx] = (
assembled_args.get(idx, "") + tc.function.arguments
)
assembled_args[idx] = assembled_args.get(idx, "") + tc.function.arguments
# delta 1 = '{"path":' + delta 2 = '"/etc/foo"}' → '{"path":"/etc/foo"}'
assert assembled_args.get(0) == '{"path":"/etc/foo"}', (
f"Assembled args for index 0 (read_file): "
f"expected '{{\"path\":\"/etc/foo\"}}', got '{assembled_args.get(0)}'"
f"Assembled args for index 0 (read_file): expected '{{\"path\":\"/etc/foo\"}}', got '{assembled_args.get(0)}'"
)
# delta 1 = '{"path":' + delta 2 = '"/tmp"}' → '{"path":"/tmp"}'
assert assembled_args.get(1) == '{"path":"/tmp"}', (
f"Assembled args for index 1 (list_dir): "
f"expected '{{\"path\":\"/tmp\"}}', got '{assembled_args.get(1)}'"
f"Assembled args for index 1 (list_dir): expected '{{\"path\":\"/tmp\"}}', got '{assembled_args.get(1)}'"
)
# 4. Stream terminates with exactly one finish event, at the final response.completed chunk
@ -2409,16 +2313,13 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
for i, r in enumerate(results)
if r is not None and r.choices and r.choices[0].finish_reason
]
assert (
len(finish_events) == 1
), f"Expected exactly 1 finish event, got {len(finish_events)}: {finish_events}"
assert len(finish_events) == 1, f"Expected exactly 1 finish event, got {len(finish_events)}: {finish_events}"
assert finish_events[0][0] == len(chunks) - 1, (
f"Finish event must be at the last chunk (index {len(chunks) - 1}), "
f"but was at index {finish_events[0][0]}"
f"Finish event must be at the last chunk (index {len(chunks) - 1}), but was at index {finish_events[0][0]}"
)
assert finish_events[0][1] == "tool_calls", (
f"Terminal finish_reason must be 'tool_calls', got '{finish_events[0][1]}'"
)
assert (
finish_events[0][1] == "tool_calls"
), f"Terminal finish_reason must be 'tool_calls', got '{finish_events[0][1]}'"
# 5. Parallel tool calls have distinct indices matching output_index (0 and 1)
# Collect indices from output_item.added chunks only (they carry the call id)
@ -2434,9 +2335,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration():
1,
}, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}"
print(
"✓ Parallel tool calls with split argument deltas stream correctly end-to-end"
)
print("✓ Parallel tool calls with split argument deltas stream correctly end-to-end")
def test_map_optional_params_preserves_reasoning_summary():
@ -2460,9 +2359,7 @@ def test_map_optional_params_preserves_reasoning_summary():
}
responses_api_request = ResponsesAPIOptionalRequestParams()
handler._map_optional_params_to_responses_api_request(
optional_params, responses_api_request
)
handler._map_optional_params_to_responses_api_request(optional_params, responses_api_request)
# Verify reasoning_effort dict with summary was fully preserved
assert "reasoning" in responses_api_request
@ -2735,9 +2632,7 @@ def test_reasoning_items_non_streaming_round_trip():
)
usage = ResponseAPIUsage(
input_tokens=10,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None),
output_tokens=20,
output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None),
total_tokens=30,
@ -2801,9 +2696,7 @@ def test_reasoning_items_non_streaming_round_trip():
assert len(result.choices) == 1
msg = result.choices[0].message
assert (
msg.reasoning_content == summary_text
), "reasoning_content should equal summary text"
assert msg.reasoning_content == summary_text, "reasoning_content should equal summary text"
assert msg.reasoning_items is not None, "reasoning_items should be set"
assert len(msg.reasoning_items) == 1
@ -2828,13 +2721,9 @@ def test_reasoning_items_non_streaming_round_trip():
# The reasoning input item must appear before the assistant message item
types = [item.get("type") for item in input_items]
assert (
"reasoning" in types
), "reasoning input item must be emitted for the assistant turn"
assert "reasoning" in types, "reasoning input item must be emitted for the assistant turn"
reasoning_input = next(
item for item in input_items if item.get("type") == "reasoning"
)
reasoning_input = next(item for item in input_items if item.get("type") == "reasoning")
assert reasoning_input["id"] == "rs_test001"
assert reasoning_input["encrypted_content"] == encrypted
assert reasoning_input["summary"][0]["text"] == summary_text
@ -2842,13 +2731,9 @@ def test_reasoning_items_non_streaming_round_trip():
# reasoning item must come before the assistant message item
reasoning_idx = types.index("reasoning")
assistant_msg_idx = next(
i
for i, item in enumerate(input_items)
if item.get("type") == "message" and item.get("role") == "assistant"
i for i, item in enumerate(input_items) if item.get("type") == "message" and item.get("role") == "assistant"
)
assert (
reasoning_idx < assistant_msg_idx
), "reasoning input item must precede the assistant message item"
assert reasoning_idx < assistant_msg_idx, "reasoning input item must precede the assistant message item"
def test_reasoning_items_streaming_emitted_on_response_completed():
@ -2861,9 +2746,7 @@ def test_reasoning_items_streaming_emitted_on_response_completed():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
encrypted = "gAAAAABpw5xyz987FAKE=="
summary_text = "**Reasoning summary**\n\nModel thought about this carefully."
@ -2907,16 +2790,14 @@ def test_reasoning_items_streaming_emitted_on_response_completed():
assert result.choices[0].finish_reason == "stop"
# reasoning_items must be on the delta
assert (
getattr(delta, "reasoning_items", None) is not None
), "reasoning_items must be present on the response.completed delta"
assert getattr(delta, "reasoning_items", None) is not None, (
"reasoning_items must be present on the response.completed delta"
)
assert len(delta.reasoning_items) == 1
ri = delta.reasoning_items[0]
assert ri["type"] == "reasoning"
assert ri["id"] == "rs_stream001"
assert (
ri["encrypted_content"] == encrypted
), "encrypted_content must be preserved in streaming"
assert ri["encrypted_content"] == encrypted, "encrypted_content must be preserved in streaming"
assert ri["summary"][0]["text"] == summary_text
@ -2943,9 +2824,7 @@ def test_streaming_function_call_tool_id_for_degenerate_call_id():
"arguments": "",
},
}
out = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
chunk
)
out = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk)
tool_calls = out.model_dump()["choices"][0]["delta"]["tool_calls"]
assert tool_calls, "expected a tool_call chunk in the streaming delta"
return tool_calls[0]["id"]
@ -2964,9 +2843,7 @@ def test_streaming_chunks_share_one_chat_completion_id():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
events = [
{"type": "response.created", "response": {"id": "resp_abc", "output": []}},
{"type": "response.output_text.delta", "delta": "Hel"},
@ -2982,12 +2859,10 @@ def test_streaming_chunks_share_one_chat_completion_id():
assert len(set(ids)) == 1, f"streamed chunks carried different ids: {ids}"
assert ids[0], "streamed chunks carried an empty id"
other_stream = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
other_stream = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
assert other_stream.chunk_parser(events[1]).id != ids[0], (
"a separate stream must get its own id, not a process-wide one"
)
assert (
other_stream.chunk_parser(events[1]).id != ids[0]
), "a separate stream must get its own id, not a process-wide one"
@pytest.mark.asyncio
@ -2998,9 +2873,7 @@ def test_streaming_chunks_share_one_chat_completion_id():
({"include_usage": True}, None),
],
)
async def test_acompletion_bridge_normalizes_stream_options_on_the_wire(
stream_options, expected_wire_stream_options
):
async def test_acompletion_bridge_normalizes_stream_options_on_the_wire(stream_options, expected_wire_stream_options):
"""include_usage must be stripped from the /v1/responses body; include_obfuscation must survive as a dict."""
from unittest.mock import AsyncMock
@ -3076,9 +2949,7 @@ def test_chunk_parser_custom_tool_call_stream_sequence():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
added = iterator.chunk_parser(
{
@ -3156,9 +3027,7 @@ def test_chunk_parser_remaps_tool_call_indices_sequentially():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
first = iterator.chunk_parser(
{
@ -3735,9 +3604,7 @@ def _make_incomplete_responses_api_response(
created_at=1760144904,
error=None,
incomplete_details=(
{"reason": incomplete_reason}
if incomplete_reason is not None or empty_incomplete_details
else None
{"reason": incomplete_reason} if incomplete_reason is not None or empty_incomplete_details else None
),
instructions=None,
metadata={},
@ -3757,13 +3624,9 @@ def _make_incomplete_responses_api_response(
truncation="disabled",
usage=ResponseAPIUsage(
input_tokens=37,
input_tokens_details=InputTokensDetails(
audio_tokens=None, cached_tokens=0, text_tokens=None
),
input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None),
output_tokens=16,
output_tokens_details=OutputTokensDetails(
reasoning_tokens=16, text_tokens=None
),
output_tokens_details=OutputTokensDetails(reasoning_tokens=16, text_tokens=None),
total_tokens=53,
cost=None,
),
@ -3813,9 +3676,7 @@ def _call_transform_response(
def test_transform_response_incomplete_reasoning_only_returns_empty_length_choice():
handler = LiteLLMResponsesTransformationHandler()
raw_response = _make_incomplete_responses_api_response(
"max_output_tokens", [_make_reasoning_only_output_item()]
)
raw_response = _make_incomplete_responses_api_response("max_output_tokens", [_make_reasoning_only_output_item()])
result = _call_transform_response(handler, raw_response)
@ -3834,9 +3695,7 @@ def test_transform_response_incomplete_reasoning_only_returns_empty_length_choic
def test_transform_response_incomplete_content_filter_maps_finish_reason():
handler = LiteLLMResponsesTransformationHandler()
raw_response = _make_incomplete_responses_api_response(
"content_filter", [_make_reasoning_only_output_item()]
)
raw_response = _make_incomplete_responses_api_response("content_filter", [_make_reasoning_only_output_item()])
result = _call_transform_response(handler, raw_response)
@ -3859,11 +3718,7 @@ def test_transform_response_completed_with_reasonless_incomplete_details_keeps_s
handler = LiteLLMResponsesTransformationHandler()
output_message = ResponseOutputMessage(
id="msg_complete",
content=[
ResponseOutputText(
annotations=[], text="full answer", type="output_text", logprobs=[]
)
],
content=[ResponseOutputText(annotations=[], text="full answer", type="output_text", logprobs=[])],
role="assistant",
status="completed",
type="message",
@ -3885,11 +3740,7 @@ def test_transform_response_incomplete_partial_text_overrides_finish_reason_to_l
handler = LiteLLMResponsesTransformationHandler()
output_message = ResponseOutputMessage(
id="msg_partial",
content=[
ResponseOutputText(
annotations=[], text="partial answer", type="output_text", logprobs=[]
)
],
content=[ResponseOutputText(annotations=[], text="partial answer", type="output_text", logprobs=[])],
role="assistant",
status="incomplete",
type="message",
@ -3911,9 +3762,7 @@ def test_response_incomplete_stream_event_emits_length_and_usage():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.incomplete",
@ -3954,9 +3803,7 @@ def test_response_incomplete_stream_event_content_filter_maps_finish_reason():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.incomplete",
@ -3978,9 +3825,7 @@ def test_response_incomplete_stream_event_without_details_defaults_to_length():
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunk = {
"type": "response.incomplete",
@ -4060,9 +3905,7 @@ def test_thinking_only_assistant_turn_still_sends_its_reasoning():
{
"role": "assistant",
"content": None,
"thinking_blocks": [
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"}
],
"thinking_blocks": [{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"}],
},
{"role": "user", "content": "Why?"},
]
@ -4088,9 +3931,7 @@ def test_stored_reasoning_items_win_over_thinking_blocks():
"summary": [{"type": "summary_text", "text": "August in Denver is dry."}],
}
],
"thinking_blocks": [
{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"}
],
"thinking_blocks": [{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"}],
},
]
@ -4560,3 +4401,76 @@ def test_every_bridged_chunk_after_response_created_carries_the_served_service_t
relayed = [iterator.chunk_parser(event).model_dump().get("service_tier") for event in events]
assert relayed == ["default"] * len(events), relayed
def test_stream_translator_emits_url_citation_annotations_exactly_once():
"""Regression for issue #43817: the Responses API delivers annotations as
response.output_text.annotation.added events, and the Chat Completions
stream bridge dropped them, so a streamed reply never carried the
url_citation that the same reply with stream=False reports on
message.annotations. The annotation must ride exactly one delta — the
terminal response.completed event repeats it in the response payload and
re-emitting it there would duplicate it for accumulating clients."""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
citation = {
"type": "url_citation",
"url": "https://a.example",
"title": "A",
"start_index": 0,
"end_index": 5,
}
reply = {
"id": "resp_1",
"object": "response",
"created_at": 0,
"status": "completed",
"model": "gpt-5-mini",
"output": [
{
"type": "message",
"id": "msg_1",
"role": "assistant",
"status": "completed",
"content": [
{
"type": "output_text",
"text": "Sunny in Paris.",
"annotations": [citation],
}
],
}
],
}
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
events = [
{"type": "response.created", "response": {**reply, "status": "in_progress", "output": []}},
{
"type": "response.output_text.delta",
"output_index": 0,
"item_id": "msg_1",
"content_index": 0,
"delta": "Sunny in Paris.",
},
{
"type": "response.output_text.annotation.added",
"output_index": 0,
"item_id": "msg_1",
"content_index": 0,
"annotation_index": 0,
"annotation": citation,
},
{"type": "response.completed", "response": reply},
]
chunks = [iterator.chunk_parser(event) for event in events]
carried = [
annotation
for chunk in chunks
for choice in chunk.choices
for annotation in (getattr(choice.delta, "annotations", None) or [])
]
assert carried == [citation]