mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix: minimize context_management polyfill threading
- Use None (not empty list) for polyfill_applied_edits when context management isn't requested, so semantics of 'feature not requested' vs 'feature requested but no edits applied' are distinct. - In the streaming iterator, only pass applied_edits to the per-chunk translator on the final (finish_reason) chunk; intermediate chunks ignore it anyway, and this makes intent explicit on both sync and async paths. Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
ac06e2f3b4
commit
16713971df
2 changed files with 14 additions and 4 deletions
|
|
@ -138,10 +138,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
if should_start_new_block:
|
||||
self._increment_content_block_index()
|
||||
|
||||
# applied_edits only needs to flow to the final message_delta
|
||||
# (when finish_reason is set); skip threading it through every
|
||||
# intermediate chunk so context_management is attached exactly
|
||||
# once, on the truly final event.
|
||||
is_final_chunk = chunk.choices[0].finish_reason is not None
|
||||
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
|
||||
response=chunk,
|
||||
current_content_block_index=self.current_content_block_index,
|
||||
applied_edits=self.applied_edits or None,
|
||||
applied_edits=self.applied_edits if is_final_chunk else None,
|
||||
)
|
||||
|
||||
if should_start_new_block and not self.sent_content_block_finish:
|
||||
|
|
@ -280,10 +285,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
if should_start_new_block:
|
||||
self._increment_content_block_index()
|
||||
|
||||
# applied_edits only needs to flow to the final message_delta
|
||||
# (when finish_reason is set); skip threading it through every
|
||||
# intermediate chunk. For the hold-and-merge path below,
|
||||
# context_management is attached directly to the merged chunk.
|
||||
is_final_chunk = chunk.choices[0].finish_reason is not None
|
||||
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
|
||||
response=chunk,
|
||||
current_content_block_index=self.current_content_block_index,
|
||||
applied_edits=self.applied_edits or None,
|
||||
applied_edits=self.applied_edits if is_final_chunk else None,
|
||||
)
|
||||
|
||||
# Check if this is a usage chunk and we have a held stop_reason chunk
|
||||
|
|
|
|||
|
|
@ -468,7 +468,7 @@ def anthropic_messages_handler(
|
|||
if _request_drop_params is not None
|
||||
else litellm.drop_params
|
||||
)
|
||||
polyfill_applied_edits: List[AppliedEdit] = []
|
||||
polyfill_applied_edits: Optional[List[AppliedEdit]] = None
|
||||
if context_management_spec and not _drop_params:
|
||||
from litellm.llms.anthropic.experimental_pass_through.context_management import (
|
||||
apply_context_management,
|
||||
|
|
@ -488,7 +488,7 @@ def anthropic_messages_handler(
|
|||
"context_management polyfill: skipping edits due to error: %s",
|
||||
e,
|
||||
)
|
||||
polyfill_applied_edits = []
|
||||
polyfill_applied_edits = None
|
||||
|
||||
return (
|
||||
LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue