diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index aa5708088b7..2a94fcb2a89 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -966,9 +966,9 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): and the ReasoningSummaryTextDeltaEvent, which is used by the responses API to emit reasoning content. It also handles emitting annotation.added events when annotations are detected in the chunk. """ - if self._cached_item_id is None and chunk.id: - self._cached_item_id = chunk.id - item_id: Final = self._cached_item_id or chunk.id + if self._cached_item_id is None: + self._cached_item_id = f"msg_{uuid.uuid4()}" + item_id: Final = self._cached_item_id # Check if this chunk has annotations first (before processing text/reasoning) # This ensures we detect and queue annotation events from the annotation chunk @@ -1003,9 +1003,12 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): ): reasoning_content: Final = chunk.choices[0].delta.reasoning_content + if self._cached_reasoning_item_id is None: + self._cached_reasoning_item_id = f"rs_{uuid.uuid4()}" + return ReasoningSummaryTextDeltaEvent( type=ResponsesAPIStreamEvents.REASONING_SUMMARY_TEXT_DELTA, - item_id=f"rs_{hash(str(reasoning_content))}", + item_id=self._cached_reasoning_item_id, output_index=0, delta=reasoning_content, ) @@ -1056,6 +1059,35 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): chat_completion_delta: Final[ChatCompletionDelta] = choice.delta return chat_completion_delta.content or "" + def _align_output_item_ids_with_streamed_ids(self, responses_api_response: ResponsesAPIResponse) -> None: + """ + Reuse the item IDs already emitted by the incremental streaming events in the + ``response.completed`` snapshot, so a streaming client that replays the snapshot + sends back the same IDs it observed mid-stream. + """ + self._set_first_output_item_id(responses_api_response, "message", self._cached_item_id) + self._set_first_output_item_id(responses_api_response, "reasoning", self._cached_reasoning_item_id) + + @staticmethod + def _set_first_output_item_id( + responses_api_response: ResponsesAPIResponse, + item_type: str, + cached_id: str | None, + ) -> None: + if cached_id is None: + return + + for item in getattr(responses_api_response, "output", None) or []: + current_type = item.get("type") if isinstance(item, dict) else getattr(item, "type", None) + if current_type != item_type: + continue + + if isinstance(item, dict): + item["id"] = cached_id + else: + item.id = cached_id + return + def _emit_response_completed_event(self, litellm_model_response: ModelResponse) -> ResponseCompletedEvent | None: if litellm_model_response: # Add cost to usage object if include_cost_in_streaming_usage is True @@ -1081,6 +1113,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): if self._cached_response_id: responses_api_response.id = self._cached_response_id + self._align_output_item_ids_with_streamed_ids(responses_api_response) + # Encode the response ID to match non-streaming behavior encoded_response: Final = ResponsesAPIRequestUtils._update_responses_api_response_id_with_model_id( responses_api_response=responses_api_response, diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 64084bfb063..cc2759358d2 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -4,6 +4,7 @@ Handles transforming from Responses API -> LiteLLM completion (Chat Completion import json import re +import uuid from collections.abc import Iterator, Mapping, Sequence from types import MappingProxyType from typing import ( @@ -2017,7 +2018,7 @@ class LiteLLMCompletionResponsesConfig: return [ GenericResponseOutputItem( type="reasoning", - id=f"rs_{hash(reasoning_content or encrypted_content)}", + id=f"rs_{uuid.uuid4()}", status=LiteLLMCompletionResponsesConfig._map_chat_completion_finish_reason_to_responses_status( choice.finish_reason ), @@ -2054,7 +2055,7 @@ class LiteLLMCompletionResponsesConfig: To Responses API format: { 'type': 'image_generation_call', - 'id': 'img_...', + 'id': 'ig_...', 'status': 'completed', 'result': 'iVBORw0...' # Pure base64 without data: prefix } @@ -2065,7 +2066,7 @@ class LiteLLMCompletionResponsesConfig: if not images: return image_generation_items - for idx, image_item in enumerate(_DICT_ITEMS_LIST_ADAPTER.validate_python(images)): + for image_item in _DICT_ITEMS_LIST_ADAPTER.validate_python(images): # Extract base64 from data URL image_url = _TEXT_ADAPTER.validate_python( _ANY_KEY_DICT_ADAPTER.validate_python(image_item.get("image_url", {})).get("url", "") @@ -2076,7 +2077,7 @@ class LiteLLMCompletionResponsesConfig: image_generation_items.append( OutputImageGenerationCall( type="image_generation_call", - id=f"{chat_completion_response.id}_img_{idx}", + id=f"ig_{uuid.uuid4()}", status=LiteLLMCompletionResponsesConfig._map_finish_reason_to_image_generation_status( choice.finish_reason ), @@ -2150,7 +2151,7 @@ class LiteLLMCompletionResponsesConfig: message_output_items.append( GenericResponseOutputItem( type="message", - id=chat_completion_response.id, + id=f"msg_{uuid.uuid4()}", status=LiteLLMCompletionResponsesConfig._map_chat_completion_finish_reason_to_responses_status( choice.finish_reason ), diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_image_generation_output.py b/tests/test_litellm/responses/litellm_completion_transformation/test_image_generation_output.py index ed7a3f63a8e..41057d49a97 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_image_generation_output.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_image_generation_output.py @@ -89,8 +89,9 @@ class TestExtractImageGenerationOutputItems: assert result[0].type == "image_generation_call" assert result[0].result == "IMG1" assert result[1].result == "IMG2" - assert result[0].id == "test_123_img_0" - assert result[1].id == "test_123_img_1" + assert result[0].id.startswith("ig_") + assert result[1].id.startswith("ig_") + assert result[0].id != result[1].id assert result[0].status == "completed" def test_returns_empty_for_no_images(self): diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 5efabed4b8d..daa732e032a 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2841,9 +2841,9 @@ class TestStreamingIDConsistency: # Verify the cached ID is set and matches assert iterator._cached_item_id is not None, "Iterator should cache the item_id" assert iterator._cached_item_id == item_id_1, "Cached ID should match event IDs" - assert ( - iterator._cached_item_id == "chatcmpl-first-id" - ), "Should use the first chunk's ID" + assert iterator._cached_item_id.startswith( + "msg_" + ), "Message item IDs must use the Responses API msg_ prefix (issue #27333)" def test_streaming_iterator_initial_events_use_cached_id(self): """ diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_response_output_item_id_prefixes.py b/tests/test_litellm/responses/litellm_completion_transformation/test_response_output_item_id_prefixes.py new file mode 100644 index 00000000000..4b24b2be90d --- /dev/null +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_response_output_item_id_prefixes.py @@ -0,0 +1,214 @@ +""" +Regression tests for the Chat Completions -> Responses API bridge item IDs. + +Bridged output items must carry Responses API ID prefixes (msg_, ig_, rs_) rather +than the upstream chatcmpl-* ID. Native OpenAI Responses rejects a replayed history +whose message item ID does not begin with "msg", and rejects an image generation +call whose ID does not begin with "ig". + +Regression test for https://github.com/BerriAI/litellm/issues/27333 +""" + +from unittest.mock import Mock + +import litellm +from litellm.responses.litellm_completion_transformation.streaming_iterator import ( + LiteLLMCompletionStreamingIterator, +) +from litellm.responses.litellm_completion_transformation.transformation import ( + LiteLLMCompletionResponsesConfig, +) +from litellm.types.utils import ( + Choices, + Delta, + Message, + ModelResponse, + ModelResponseStream, + StreamingChoices, + Usage, +) + +CHAT_COMPLETION_ID = "chatcmpl-dfa2da3a-1586-4ff7-b64e-f59c692a5d11" + + +def _make_chat_completion_response(**overrides) -> ModelResponse: + defaults = dict( + id=CHAT_COMPLETION_ID, + created=1717000000, + model="claude-sonnet-4-5", + object="chat.completion", + choices=[ + Choices( + index=0, + finish_reason="stop", + message=Message(role="assistant", content="apple"), + ) + ], + usage=Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15), + ) + defaults.update(overrides) + return ModelResponse(**defaults) + + +def _transform(chat_completion_response): + return LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response( + request_input="Say the single word: apple", + responses_api_request={}, + chat_completion_response=chat_completion_response, + ) + + +def _output_items_of_type(response, item_type): + return [item for item in response.output if getattr(item, "type", None) == item_type] + + +class TestMessageOutputItemIds: + def test_message_item_id_uses_msg_prefix(self): + response = _transform(_make_chat_completion_response()) + + message_items = _output_items_of_type(response, "message") + assert len(message_items) == 1 + assert message_items[0].id.startswith("msg_") + + def test_message_item_id_does_not_leak_chat_completion_id(self): + response = _transform(_make_chat_completion_response()) + + for item in _output_items_of_type(response, "message"): + assert item.id != CHAT_COMPLETION_ID + assert not item.id.startswith("chatcmpl-") + + def test_message_item_ids_are_unique_across_responses(self): + first = _transform(_make_chat_completion_response()) + second = _transform(_make_chat_completion_response()) + + first_id = _output_items_of_type(first, "message")[0].id + second_id = _output_items_of_type(second, "message")[0].id + assert first_id != second_id + + +class TestImageGenerationOutputItemIds: + def _make_choice_with_images(self, count): + message = Mock(spec=Message) + message.images = [ + {"image_url": {"url": f"data:image/png;base64,IMG{idx}"}} for idx in range(count) + ] + choice = Mock(spec=Choices) + choice.message = message + choice.finish_reason = "stop" + return choice + + def test_image_generation_item_id_uses_ig_prefix(self): + items = LiteLLMCompletionResponsesConfig._extract_image_generation_output_items( + chat_completion_response=_make_chat_completion_response(), + choice=self._make_choice_with_images(2), + ) + + assert len(items) == 2 + for item in items: + assert item.id.startswith("ig_") + assert "chatcmpl-" not in item.id + assert "_img_" not in item.id + + def test_image_generation_item_ids_are_unique(self): + items = LiteLLMCompletionResponsesConfig._extract_image_generation_output_items( + chat_completion_response=_make_chat_completion_response(), + choice=self._make_choice_with_images(3), + ) + + assert len({item.id for item in items}) == 3 + + +class TestReasoningOutputItemIds: + def _reasoning_items(self): + message = Message(role="assistant", content="apple") + message.reasoning_content = "thinking about fruit" + choice = Choices(index=0, finish_reason="stop", message=message) + return LiteLLMCompletionResponsesConfig._extract_reasoning_output_items( + chat_completion_response=_make_chat_completion_response(), + choices=[choice], + ) + + def test_reasoning_item_id_uses_rs_prefix(self): + items = self._reasoning_items() + + assert len(items) == 1 + assert items[0].id.startswith("rs_") + + def test_reasoning_item_id_is_not_a_salted_hash(self): + item_id = self._reasoning_items()[0].id + + suffix = item_id.removeprefix("rs_") + assert not suffix.lstrip("-").isdigit() + assert not suffix.startswith("-") + + +class TestStreamingItemIdConsistency: + def _make_iterator(self): + mock_stream_wrapper = Mock(spec=litellm.CustomStreamWrapper) + mock_stream_wrapper.logging_obj = Mock() + return LiteLLMCompletionStreamingIterator( + model="anthropic/claude-sonnet-4-5", + litellm_custom_stream_wrapper=mock_stream_wrapper, + request_input="Say the single word: apple", + responses_api_request={}, + custom_llm_provider="anthropic", + ) + + def _make_chunk(self, chunk_id, content, finish_reason=None): + return ModelResponseStream( + id=chunk_id, + choices=[ + StreamingChoices( + index=0, + delta=Delta(content=content, role="assistant"), + finish_reason=finish_reason, + ) + ], + created=1717000000, + model="claude-sonnet-4-5", + object="chat.completion.chunk", + ) + + def test_incremental_item_id_uses_msg_prefix(self): + iterator = self._make_iterator() + + event = iterator._transform_chat_completion_chunk_to_response_api_chunk( + self._make_chunk(CHAT_COMPLETION_ID, "apple") + ) + + assert event is not None + assert event.item_id.startswith("msg_") + assert event.item_id != CHAT_COMPLETION_ID + + def test_completed_snapshot_reuses_streamed_item_id(self): + iterator = self._make_iterator() + + streamed_event = iterator._transform_chat_completion_chunk_to_response_api_chunk( + self._make_chunk(CHAT_COMPLETION_ID, "apple") + ) + assert streamed_event is not None + streamed_item_id = streamed_event.item_id + + completed_event = iterator._emit_response_completed_event( + _make_chat_completion_response() + ) + + assert completed_event is not None + message_items = _output_items_of_type(completed_event.response, "message") + assert len(message_items) == 1 + assert message_items[0].id == streamed_item_id + + def test_completed_snapshot_item_id_is_replayable(self): + iterator = self._make_iterator() + iterator._transform_chat_completion_chunk_to_response_api_chunk( + self._make_chunk(CHAT_COMPLETION_ID, "apple") + ) + + completed_event = iterator._emit_response_completed_event( + _make_chat_completion_response() + ) + + assert completed_event is not None + for item in _output_items_of_type(completed_event.response, "message"): + assert item.id.startswith("msg_") + assert not item.id.startswith("chatcmpl-")