From 50fff0d7cf7c633670d9b1b5f1a6cbaa361578f9 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 18 May 2026 09:15:40 +0000 Subject: [PATCH] fix(interactions): stabilize streaming bridge schema, dict aliasing, and lost first delta - Capture use_legacy_interactions_schema once at iterator construction so all events emitted by a single stream use a consistent schema, even if the global flag is mutated mid-stream. - Check for the buffered interaction.complete/completed event before the finished check in __next__/__anext__ so the final completion event (which carries the full collected text in steps) is not dropped after self.finished is set. - Copy text content entries before appending to both outputs and the steps content list to avoid shared mutable dict aliasing between the two response fields. Co-authored-by: Yassin Kortam --- .../streaming_iterator.py | 32 +++++++++++-------- .../transformation.py | 11 ++++--- 2 files changed, 24 insertions(+), 19 deletions(-) diff --git a/litellm/interactions/litellm_responses_transformation/streaming_iterator.py b/litellm/interactions/litellm_responses_transformation/streaming_iterator.py index 9c18b511605..0d7e40cd420 100644 --- a/litellm/interactions/litellm_responses_transformation/streaming_iterator.py +++ b/litellm/interactions/litellm_responses_transformation/streaming_iterator.py @@ -47,6 +47,8 @@ class LiteLLMResponsesInteractionsStreamingIterator: custom_llm_provider: Optional[str] = None, litellm_metadata: Optional[Dict[str, Any]] = None, ): + import litellm + self.model = model self.responses_stream_iterator = litellm_custom_stream_wrapper self.request_input = request_input @@ -57,12 +59,10 @@ class LiteLLMResponsesInteractionsStreamingIterator: self.collected_text = "" self.sent_interaction_start = False self.sent_content_start = False - - @property - def _use_legacy(self) -> bool: - import litellm - - return litellm.use_legacy_interactions_schema + # Capture the schema flag once at construction time so all events + # emitted by this stream use a consistent schema, even if the global + # flag is mutated mid-stream (e.g. by a config reload). + self._use_legacy: bool = litellm.use_legacy_interactions_schema def _transform_responses_chunk_to_interactions_chunk( self, @@ -201,10 +201,9 @@ class LiteLLMResponsesInteractionsStreamingIterator: def __next__(self) -> InteractionsAPIStreamingResponse: """Get next chunk in sync mode.""" - if self.finished: - raise StopIteration - - # Check if we have a pending interaction.complete to send + # Check for a pending interaction.complete/completed event BEFORE the + # finished check — otherwise the buffered completion event (which + # carries the full text) would be dropped after `self.finished` is set. if hasattr(self, "_pending_interaction_complete"): pending: InteractionsAPIStreamingResponse = getattr( self, "_pending_interaction_complete" @@ -212,6 +211,9 @@ class LiteLLMResponsesInteractionsStreamingIterator: delattr(self, "_pending_interaction_complete") return pending + if self.finished: + raise StopIteration + # Use a loop instead of recursion to avoid stack overflow sync_iterator = cast( SyncResponsesAPIStreamingIterator, self.responses_stream_iterator @@ -287,10 +289,9 @@ class LiteLLMResponsesInteractionsStreamingIterator: async def __anext__(self) -> InteractionsAPIStreamingResponse: """Get next chunk in async mode.""" - if self.finished: - raise StopAsyncIteration - - # Check if we have a pending interaction.complete to send + # Check for a pending interaction.complete/completed event BEFORE the + # finished check — otherwise the buffered completion event (which + # carries the full text) would be dropped after `self.finished` is set. if hasattr(self, "_pending_interaction_complete"): pending: InteractionsAPIStreamingResponse = getattr( self, "_pending_interaction_complete" @@ -298,6 +299,9 @@ class LiteLLMResponsesInteractionsStreamingIterator: delattr(self, "_pending_interaction_complete") return pending + if self.finished: + raise StopAsyncIteration + # Use a loop instead of recursion to avoid stack overflow async_iterator = cast( ResponsesAPIStreamingIterator, self.responses_stream_iterator diff --git a/litellm/interactions/litellm_responses_transformation/transformation.py b/litellm/interactions/litellm_responses_transformation/transformation.py index dbc50cb0dec..173d4ca8764 100644 --- a/litellm/interactions/litellm_responses_transformation/transformation.py +++ b/litellm/interactions/litellm_responses_transformation/transformation.py @@ -240,15 +240,16 @@ class LiteLLMResponsesInteractionsConfig: # Check if content_item has text attribute text = getattr(content_item, "text", None) if text is not None: - text_entry = {"type": "text", "text": text} - outputs.append(text_entry) - model_output_contents.append(text_entry) + # Use independent dict instances so mutations to one + # of `outputs` / `steps` don't leak into the other. + outputs.append({"type": "text", "text": text}) + model_output_contents.append({"type": "text", "text": text}) elif ( isinstance(content_item, dict) and content_item.get("type") == "text" ): - outputs.append(content_item) - model_output_contents.append(content_item) + outputs.append({**content_item}) + model_output_contents.append({**content_item}) if model_output_contents: steps.append( {