mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(anthropic): plug content drop, compaction SSE shape, and compaction leak
- Sync streaming __next__ no longer drops a buffered holding_chunk when
the usage-merge path has already fired. Restoring the prior unconditional
flush behavior preserves provider-emitted content (the SSE-ordering nit
of a trailing content delta is preferable to silent content loss).
- compaction content_block_start now carries the full block shape
({"type": "compaction", "content": ""}) to match the text-block
pattern and Anthropic's native streaming shape, so clients that key off
content_block_start see the field.
- apply_compact_20260112 now slices around / strips compaction blocks
before the opt-in gate check. Previously, when summary_model was not
configured the editor returned the raw messages, leaking Anthropic-only
compaction content blocks to non-Anthropic providers that reject them.
Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
6527ea94c9
commit
8551e939d5
2 changed files with 29 additions and 15 deletions
|
|
@ -175,7 +175,12 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
{
|
||||
"type": "content_block_start",
|
||||
"index": compaction_index,
|
||||
"content_block": {"type": "compaction"},
|
||||
# Mirror the text-block shape ({"type": "text", "text": ""}):
|
||||
# send an empty ``content`` field so clients that introspect
|
||||
# ``content_block_start`` see the full block schema. The
|
||||
# actual summary text arrives via the ``content_block_delta``
|
||||
# below.
|
||||
"content_block": {"type": "compaction", "content": ""},
|
||||
}
|
||||
)
|
||||
self.chunk_queue.append(
|
||||
|
|
@ -394,9 +399,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
)
|
||||
self.holding_stop_reason_chunk = None
|
||||
|
||||
if self.holding_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.holding_chunk = None
|
||||
# Always flush any buffered content delta, even when usage has
|
||||
# already been merged + emitted: dropping it would silently lose
|
||||
# provider-emitted content, which is worse than the SSE ordering
|
||||
# nit of trailing a content chunk after the final message_delta
|
||||
# (the prior sync ``__next__`` behavior).
|
||||
if self.holding_chunk is not None:
|
||||
self.chunk_queue.append(self.holding_chunk)
|
||||
self.holding_chunk = None
|
||||
|
||||
if not self.sent_last_message:
|
||||
self.sent_last_message = True
|
||||
|
|
|
|||
|
|
@ -486,17 +486,10 @@ async def apply_compact_20260112(
|
|||
if warnings:
|
||||
applied["warnings"] = warnings
|
||||
|
||||
# Opt-in gate: no summary model configured → no-op.
|
||||
summary_model = _read_summary_model_setting()
|
||||
if summary_model is None:
|
||||
applied["error"] = "summary_model_not_configured"
|
||||
return PolyfillResult(
|
||||
messages=messages,
|
||||
system=system,
|
||||
applied_edits=[applied],
|
||||
)
|
||||
|
||||
# Phase A: slice around any existing compaction block.
|
||||
# Phase A: slice around any existing compaction block. Runs before the
|
||||
# opt-in gate below so that even when summarization is disabled we still
|
||||
# strip Anthropic-only ``compaction`` blocks from messages going to
|
||||
# non-Anthropic backends (which would reject them).
|
||||
effective_messages, prior_compaction_block = _slice_around_compaction_block(
|
||||
messages
|
||||
)
|
||||
|
|
@ -513,6 +506,17 @@ async def apply_compact_20260112(
|
|||
|
||||
downstream_messages = _strip_compaction_blocks(effective_messages)
|
||||
|
||||
# Opt-in gate: no summary model configured → no-op (but still return the
|
||||
# Phase A-sliced/stripped messages so compaction blocks don't leak).
|
||||
summary_model = _read_summary_model_setting()
|
||||
if summary_model is None:
|
||||
applied["error"] = "summary_model_not_configured"
|
||||
return PolyfillResult(
|
||||
messages=downstream_messages,
|
||||
system=augmented_system,
|
||||
applied_edits=[applied],
|
||||
)
|
||||
|
||||
# Phase B: threshold check.
|
||||
try:
|
||||
current_tokens = _count_effective_tokens(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue