diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index bec40c8c7a5..2eca78b5519 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -175,7 +175,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): { "type": "content_block_start", "index": compaction_index, - "content_block": {"type": "compaction"}, + # Mirror the text-block shape ({"type": "text", "text": ""}): + # send an empty ``content`` field so clients that introspect + # ``content_block_start`` see the full block schema. The + # actual summary text arrives via the ``content_block_delta`` + # below. + "content_block": {"type": "compaction", "content": ""}, } ) self.chunk_queue.append( @@ -394,9 +399,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): ) self.holding_stop_reason_chunk = None - if self.holding_chunk is not None: - self.chunk_queue.append(self.holding_chunk) - self.holding_chunk = None + # Always flush any buffered content delta, even when usage has + # already been merged + emitted: dropping it would silently lose + # provider-emitted content, which is worse than the SSE ordering + # nit of trailing a content chunk after the final message_delta + # (the prior sync ``__next__`` behavior). + if self.holding_chunk is not None: + self.chunk_queue.append(self.holding_chunk) + self.holding_chunk = None if not self.sent_last_message: self.sent_last_message = True diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index d1ef9ebc15e..c23461bcd9b 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -486,17 +486,10 @@ async def apply_compact_20260112( if warnings: applied["warnings"] = warnings - # Opt-in gate: no summary model configured → no-op. - summary_model = _read_summary_model_setting() - if summary_model is None: - applied["error"] = "summary_model_not_configured" - return PolyfillResult( - messages=messages, - system=system, - applied_edits=[applied], - ) - - # Phase A: slice around any existing compaction block. + # Phase A: slice around any existing compaction block. Runs before the + # opt-in gate below so that even when summarization is disabled we still + # strip Anthropic-only ``compaction`` blocks from messages going to + # non-Anthropic backends (which would reject them). effective_messages, prior_compaction_block = _slice_around_compaction_block( messages ) @@ -513,6 +506,17 @@ async def apply_compact_20260112( downstream_messages = _strip_compaction_blocks(effective_messages) + # Opt-in gate: no summary model configured → no-op (but still return the + # Phase A-sliced/stripped messages so compaction blocks don't leak). + summary_model = _read_summary_model_setting() + if summary_model is None: + applied["error"] = "summary_model_not_configured" + return PolyfillResult( + messages=downstream_messages, + system=augmented_system, + applied_edits=[applied], + ) + # Phase B: threshold check. try: current_tokens = _count_effective_tokens(