refactor(responses): build split chunks without mutating copies

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-08-09 01:48:25 +00:00
parent 2bd8020d83
commit b9bf5d82db
2 changed files with 19 additions and 12 deletions

View file

@ -150,14 +150,17 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
if not (getattr(delta, "reasoning_content", None) and delta.content):
return (chunk,)
reasoning_part: Final = chunk.model_copy(deep=True)
reasoning_part.choices[0].delta.content = None
reasoning_part.choices[0].finish_reason = None
content_part: Final = chunk.model_copy(deep=True)
content_part.choices[0].delta.reasoning_content = None
content_part.choices[0].delta.thinking_blocks = None
return (reasoning_part, content_part)
choice: Final = chunk.choices[0]
reasoning_choice: Final = choice.model_copy(
update={"delta": delta.model_copy(update={"content": None}), "finish_reason": None}
)
content_choice: Final = choice.model_copy(
update={"delta": delta.model_copy(update={"reasoning_content": None, "thinking_blocks": None})}
)
return (
chunk.model_copy(update={"choices": [reasoning_choice]}),
chunk.model_copy(update={"choices": [content_choice]}),
)
def _queue_reasoning_lifecycle_events(self, chunk: ModelResponseStream) -> None:
if not self._reasoning_active or self._reasoning_done_emitted:

View file

@ -270,7 +270,7 @@ class TestReasoningContentFinalResponse:
class _FakeChatCompletionStream:
"""Minimal CustomStreamWrapper stand-in that replays pre-built chunks."""
def __init__(self, chunks):
def __init__(self, chunks: list[ModelResponseStream]):
self._chunks = list(chunks)
self.logging_obj = SimpleNamespace(
_response_cost_calculator=lambda result: 0.0,
@ -294,8 +294,12 @@ class _FakeChatCompletionStream:
return self._chunks.pop(0)
def _reasoning_and_text_chunks():
def make(content=None, reasoning=None, finish_reason=None):
def _reasoning_and_text_chunks() -> list[ModelResponseStream]:
def make(
content: str | None = None,
reasoning: str | None = None,
finish_reason: str | None = None,
) -> ModelResponseStream:
return ModelResponseStream(
id="chatcmpl-combined",
created=1234567890,
@ -320,7 +324,7 @@ def _reasoning_and_text_chunks():
]
def _new_iterator(chunks):
def _new_iterator(chunks: list[ModelResponseStream]) -> LiteLLMCompletionStreamingIterator:
return LiteLLMCompletionStreamingIterator(
model="test-model",
litellm_custom_stream_wrapper=_FakeChatCompletionStream(chunks),