From 8551e939d582ab6fa7c55d7f20afe092325c1611 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 27 May 2026 13:54:11 +0000 Subject: [PATCH] fix(anthropic): plug content drop, compaction SSE shape, and compaction leak - Sync streaming __next__ no longer drops a buffered holding_chunk when the usage-merge path has already fired. Restoring the prior unconditional flush behavior preserves provider-emitted content (the SSE-ordering nit of a trailing content delta is preferable to silent content loss). - compaction content_block_start now carries the full block shape ({"type": "compaction", "content": ""}) to match the text-block pattern and Anthropic's native streaming shape, so clients that key off content_block_start see the field. - apply_compact_20260112 now slices around / strips compaction blocks before the opt-in gate check. Previously, when summary_model was not configured the editor returned the raw messages, leaking Anthropic-only compaction content blocks to non-Anthropic providers that reject them. Co-authored-by: Yassin Kortam --- .../adapters/streaming_iterator.py | 18 ++++++++++--- .../context_management/editors/compact.py | 26 +++++++++++-------- 2 files changed, 29 insertions(+), 15 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index bec40c8c7a5..2eca78b5519 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -175,7 +175,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): { "type": "content_block_start", "index": compaction_index, - "content_block": {"type": "compaction"}, + # Mirror the text-block shape ({"type": "text", "text": ""}): + # send an empty ``content`` field so clients that introspect + # ``content_block_start`` see the full block schema. The + # actual summary text arrives via the ``content_block_delta`` + # below. + "content_block": {"type": "compaction", "content": ""}, } ) self.chunk_queue.append( @@ -394,9 +399,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): ) self.holding_stop_reason_chunk = None - if self.holding_chunk is not None: - self.chunk_queue.append(self.holding_chunk) - self.holding_chunk = None + # Always flush any buffered content delta, even when usage has + # already been merged + emitted: dropping it would silently lose + # provider-emitted content, which is worse than the SSE ordering + # nit of trailing a content chunk after the final message_delta + # (the prior sync ``__next__`` behavior). + if self.holding_chunk is not None: + self.chunk_queue.append(self.holding_chunk) + self.holding_chunk = None if not self.sent_last_message: self.sent_last_message = True diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index d1ef9ebc15e..c23461bcd9b 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -486,17 +486,10 @@ async def apply_compact_20260112( if warnings: applied["warnings"] = warnings - # Opt-in gate: no summary model configured → no-op. - summary_model = _read_summary_model_setting() - if summary_model is None: - applied["error"] = "summary_model_not_configured" - return PolyfillResult( - messages=messages, - system=system, - applied_edits=[applied], - ) - - # Phase A: slice around any existing compaction block. + # Phase A: slice around any existing compaction block. Runs before the + # opt-in gate below so that even when summarization is disabled we still + # strip Anthropic-only ``compaction`` blocks from messages going to + # non-Anthropic backends (which would reject them). effective_messages, prior_compaction_block = _slice_around_compaction_block( messages ) @@ -513,6 +506,17 @@ async def apply_compact_20260112( downstream_messages = _strip_compaction_blocks(effective_messages) + # Opt-in gate: no summary model configured → no-op (but still return the + # Phase A-sliced/stripped messages so compaction blocks don't leak). + summary_model = _read_summary_model_setting() + if summary_model is None: + applied["error"] = "summary_model_not_configured" + return PolyfillResult( + messages=downstream_messages, + system=augmented_system, + applied_edits=[applied], + ) + # Phase B: threshold check. try: current_tokens = _count_effective_tokens(