mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
refactor(responses): build split chunks without mutating copies
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
2bd8020d83
commit
b9bf5d82db
2 changed files with 19 additions and 12 deletions
|
|
@ -150,14 +150,17 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
|
|||
if not (getattr(delta, "reasoning_content", None) and delta.content):
|
||||
return (chunk,)
|
||||
|
||||
reasoning_part: Final = chunk.model_copy(deep=True)
|
||||
reasoning_part.choices[0].delta.content = None
|
||||
reasoning_part.choices[0].finish_reason = None
|
||||
|
||||
content_part: Final = chunk.model_copy(deep=True)
|
||||
content_part.choices[0].delta.reasoning_content = None
|
||||
content_part.choices[0].delta.thinking_blocks = None
|
||||
return (reasoning_part, content_part)
|
||||
choice: Final = chunk.choices[0]
|
||||
reasoning_choice: Final = choice.model_copy(
|
||||
update={"delta": delta.model_copy(update={"content": None}), "finish_reason": None}
|
||||
)
|
||||
content_choice: Final = choice.model_copy(
|
||||
update={"delta": delta.model_copy(update={"reasoning_content": None, "thinking_blocks": None})}
|
||||
)
|
||||
return (
|
||||
chunk.model_copy(update={"choices": [reasoning_choice]}),
|
||||
chunk.model_copy(update={"choices": [content_choice]}),
|
||||
)
|
||||
|
||||
def _queue_reasoning_lifecycle_events(self, chunk: ModelResponseStream) -> None:
|
||||
if not self._reasoning_active or self._reasoning_done_emitted:
|
||||
|
|
|
|||
|
|
@ -270,7 +270,7 @@ class TestReasoningContentFinalResponse:
|
|||
class _FakeChatCompletionStream:
|
||||
"""Minimal CustomStreamWrapper stand-in that replays pre-built chunks."""
|
||||
|
||||
def __init__(self, chunks):
|
||||
def __init__(self, chunks: list[ModelResponseStream]):
|
||||
self._chunks = list(chunks)
|
||||
self.logging_obj = SimpleNamespace(
|
||||
_response_cost_calculator=lambda result: 0.0,
|
||||
|
|
@ -294,8 +294,12 @@ class _FakeChatCompletionStream:
|
|||
return self._chunks.pop(0)
|
||||
|
||||
|
||||
def _reasoning_and_text_chunks():
|
||||
def make(content=None, reasoning=None, finish_reason=None):
|
||||
def _reasoning_and_text_chunks() -> list[ModelResponseStream]:
|
||||
def make(
|
||||
content: str | None = None,
|
||||
reasoning: str | None = None,
|
||||
finish_reason: str | None = None,
|
||||
) -> ModelResponseStream:
|
||||
return ModelResponseStream(
|
||||
id="chatcmpl-combined",
|
||||
created=1234567890,
|
||||
|
|
@ -320,7 +324,7 @@ def _reasoning_and_text_chunks():
|
|||
]
|
||||
|
||||
|
||||
def _new_iterator(chunks):
|
||||
def _new_iterator(chunks: list[ModelResponseStream]) -> LiteLLMCompletionStreamingIterator:
|
||||
return LiteLLMCompletionStreamingIterator(
|
||||
model="test-model",
|
||||
litellm_custom_stream_wrapper=_FakeChatCompletionStream(chunks),
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue