From 16713971df4a3c5da84e01efdd2d90909a4e4c7d Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 25 May 2026 14:05:14 +0000 Subject: [PATCH] fix: minimize context_management polyfill threading - Use None (not empty list) for polyfill_applied_edits when context management isn't requested, so semantics of 'feature not requested' vs 'feature requested but no edits applied' are distinct. - In the streaming iterator, only pass applied_edits to the per-chunk translator on the final (finish_reason) chunk; intermediate chunks ignore it anyway, and this makes intent explicit on both sync and async paths. Co-authored-by: Yassin Kortam --- .../adapters/streaming_iterator.py | 14 ++++++++++++-- .../experimental_pass_through/messages/handler.py | 4 ++-- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index 468283f804c..56154ee04dc 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -138,10 +138,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): if should_start_new_block: self._increment_content_block_index() + # applied_edits only needs to flow to the final message_delta + # (when finish_reason is set); skip threading it through every + # intermediate chunk so context_management is attached exactly + # once, on the truly final event. + is_final_chunk = chunk.choices[0].finish_reason is not None processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( response=chunk, current_content_block_index=self.current_content_block_index, - applied_edits=self.applied_edits or None, + applied_edits=self.applied_edits if is_final_chunk else None, ) if should_start_new_block and not self.sent_content_block_finish: @@ -280,10 +285,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): if should_start_new_block: self._increment_content_block_index() + # applied_edits only needs to flow to the final message_delta + # (when finish_reason is set); skip threading it through every + # intermediate chunk. For the hold-and-merge path below, + # context_management is attached directly to the merged chunk. + is_final_chunk = chunk.choices[0].finish_reason is not None processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic( response=chunk, current_content_block_index=self.current_content_block_index, - applied_edits=self.applied_edits or None, + applied_edits=self.applied_edits if is_final_chunk else None, ) # Check if this is a usage chunk and we have a held stop_reason chunk diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 0146362f979..2551353a049 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -468,7 +468,7 @@ def anthropic_messages_handler( if _request_drop_params is not None else litellm.drop_params ) - polyfill_applied_edits: List[AppliedEdit] = [] + polyfill_applied_edits: Optional[List[AppliedEdit]] = None if context_management_spec and not _drop_params: from litellm.llms.anthropic.experimental_pass_through.context_management import ( apply_context_management, @@ -488,7 +488,7 @@ def anthropic_messages_handler( "context_management polyfill: skipping edits due to error: %s", e, ) - polyfill_applied_edits = [] + polyfill_applied_edits = None return ( LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(