fix: minimize context_management polyfill threading

- Use None (not empty list) for polyfill_applied_edits when context
  management isn't requested, so semantics of 'feature not requested'
  vs 'feature requested but no edits applied' are distinct.
- In the streaming iterator, only pass applied_edits to the per-chunk
  translator on the final (finish_reason) chunk; intermediate chunks
  ignore it anyway, and this makes intent explicit on both sync and
  async paths.

Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
Cursor Agent 2026-05-25 14:05:14 +00:00
parent ac06e2f3b4
commit 16713971df
No known key found for this signature in database
2 changed files with 14 additions and 4 deletions

View file

@ -138,10 +138,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
if should_start_new_block:
self._increment_content_block_index()
# applied_edits only needs to flow to the final message_delta
# (when finish_reason is set); skip threading it through every
# intermediate chunk so context_management is attached exactly
# once, on the truly final event.
is_final_chunk = chunk.choices[0].finish_reason is not None
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk,
current_content_block_index=self.current_content_block_index,
applied_edits=self.applied_edits or None,
applied_edits=self.applied_edits if is_final_chunk else None,
)
if should_start_new_block and not self.sent_content_block_finish:
@ -280,10 +285,15 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
if should_start_new_block:
self._increment_content_block_index()
# applied_edits only needs to flow to the final message_delta
# (when finish_reason is set); skip threading it through every
# intermediate chunk. For the hold-and-merge path below,
# context_management is attached directly to the merged chunk.
is_final_chunk = chunk.choices[0].finish_reason is not None
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk,
current_content_block_index=self.current_content_block_index,
applied_edits=self.applied_edits or None,
applied_edits=self.applied_edits if is_final_chunk else None,
)
# Check if this is a usage chunk and we have a held stop_reason chunk

View file

@ -468,7 +468,7 @@ def anthropic_messages_handler(
if _request_drop_params is not None
else litellm.drop_params
)
polyfill_applied_edits: List[AppliedEdit] = []
polyfill_applied_edits: Optional[List[AppliedEdit]] = None
if context_management_spec and not _drop_params:
from litellm.llms.anthropic.experimental_pass_through.context_management import (
apply_context_management,
@ -488,7 +488,7 @@ def anthropic_messages_handler(
"context_management polyfill: skipping edits due to error: %s",
e,
)
polyfill_applied_edits = []
polyfill_applied_edits = None
return (
LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(