From b9bf5d82db6720144b1f69b6b8a343b2588411a9 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sun, 9 Aug 2026 01:48:25 +0000 Subject: [PATCH] refactor(responses): build split chunks without mutating copies Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 19 +++++++++++-------- .../test_reasoning_content_transformation.py | 12 ++++++++---- 2 files changed, 19 insertions(+), 12 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index f88883586ab..8ff8293f21b 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -150,14 +150,17 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): if not (getattr(delta, "reasoning_content", None) and delta.content): return (chunk,) - reasoning_part: Final = chunk.model_copy(deep=True) - reasoning_part.choices[0].delta.content = None - reasoning_part.choices[0].finish_reason = None - - content_part: Final = chunk.model_copy(deep=True) - content_part.choices[0].delta.reasoning_content = None - content_part.choices[0].delta.thinking_blocks = None - return (reasoning_part, content_part) + choice: Final = chunk.choices[0] + reasoning_choice: Final = choice.model_copy( + update={"delta": delta.model_copy(update={"content": None}), "finish_reason": None} + ) + content_choice: Final = choice.model_copy( + update={"delta": delta.model_copy(update={"reasoning_content": None, "thinking_blocks": None})} + ) + return ( + chunk.model_copy(update={"choices": [reasoning_choice]}), + chunk.model_copy(update={"choices": [content_choice]}), + ) def _queue_reasoning_lifecycle_events(self, chunk: ModelResponseStream) -> None: if not self._reasoning_active or self._reasoning_done_emitted: diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_reasoning_content_transformation.py b/tests/test_litellm/responses/litellm_completion_transformation/test_reasoning_content_transformation.py index 50a4707555f..14b6f9c3d1f 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_reasoning_content_transformation.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_reasoning_content_transformation.py @@ -270,7 +270,7 @@ class TestReasoningContentFinalResponse: class _FakeChatCompletionStream: """Minimal CustomStreamWrapper stand-in that replays pre-built chunks.""" - def __init__(self, chunks): + def __init__(self, chunks: list[ModelResponseStream]): self._chunks = list(chunks) self.logging_obj = SimpleNamespace( _response_cost_calculator=lambda result: 0.0, @@ -294,8 +294,12 @@ class _FakeChatCompletionStream: return self._chunks.pop(0) -def _reasoning_and_text_chunks(): - def make(content=None, reasoning=None, finish_reason=None): +def _reasoning_and_text_chunks() -> list[ModelResponseStream]: + def make( + content: str | None = None, + reasoning: str | None = None, + finish_reason: str | None = None, + ) -> ModelResponseStream: return ModelResponseStream( id="chatcmpl-combined", created=1234567890, @@ -320,7 +324,7 @@ def _reasoning_and_text_chunks(): ] -def _new_iterator(chunks): +def _new_iterator(chunks: list[ModelResponseStream]) -> LiteLLMCompletionStreamingIterator: return LiteLLMCompletionStreamingIterator( model="test-model", litellm_custom_stream_wrapper=_FakeChatCompletionStream(chunks),