diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index a85427a9180..3e5bd3d363a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -626,9 +626,11 @@ def apply_client_compaction_block_history( """Honor client-sent compaction blocks without a ``compact_20260112`` edit. When the request omits ``context_management`` but the message history already - contains a ``compaction`` content block (e.g. Claude Code client-side compaction), - apply the same slice-only forwarding as the under-threshold path: summary on - ``system``, latest user question only on the main call. + contains a ``compaction`` content block (e.g. Claude Code client-side + compaction), apply the same slice-only forwarding as the under-threshold + path: the prior summary is prepended to ``system`` and the post-compaction + tail is forwarded unchanged (with compaction blocks stripped) so recent + turns the summary does not cover are preserved. """ effective_messages, prior_compaction_block = _slice_around_compaction_block( messages @@ -650,7 +652,12 @@ def apply_client_compaction_block_history( len(prior_summary_text), ) - downstream_messages = _select_last_user_question(effective_messages) + # Post-compaction turns are recent context the prior summary does not cover, + # so forward them unchanged. Only fall back to the last user question if the + # strip leaves the downstream call with nothing to answer. + downstream_messages = _strip_compaction_blocks(effective_messages) + if not downstream_messages: + downstream_messages = _select_last_user_question(effective_messages) return PolyfillResult( messages=downstream_messages, @@ -750,14 +757,13 @@ async def apply_compact_20260112( # noqa: PLR0915 ) if current_tokens <= trigger_tokens: - # Slice-only path: prior context lives in ``augmented_system`` (the - # compaction summary prefix). The main model call must not re-send stale - # assistant turns from the post-compaction tail — only the latest user - # question, matching the full-summary path below. - if prior_compaction_block is not None: - downstream_messages = _select_last_user_question(effective_messages) - elif not downstream_messages: - # No compaction checkpoint: only substitute when strip left nothing. + # Slice-only path: the prior compaction summary already lives in + # ``augmented_system``. Post-compaction turns are recent context the + # summary does not cover, so forward ``downstream_messages`` (the + # post-compaction tail with compaction blocks stripped) unchanged. + # Only fall back to the last user question when the strip leaves + # nothing for the downstream call to answer. + if not downstream_messages: downstream_messages = _select_last_user_question(effective_messages) return PolyfillResult( messages=downstream_messages, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index 580e3d37af8..4407e13d75f 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -358,7 +358,13 @@ async def test_opt_in_gating_no_summary_model_configured(): def test_client_compaction_block_history_without_context_management(): - """Compaction in messages alone triggers slice-only forwarding.""" + """Compaction in messages alone triggers slice-only forwarding. + + The prior summary is prepended to ``system``; the post-compaction tail is + forwarded unchanged so the model sees the recent turns the summary does + not cover. Compaction blocks themselves are stripped from messages so + non-Anthropic backends don't reject them. + """ messages = _messages_with_compaction("prior summary text") result = apply_client_compaction_block_history(messages=messages, system=None) @@ -368,9 +374,15 @@ def test_client_compaction_block_history_without_context_management(): assert "prior summary text" in str(result.system) assert result.compaction_block is None assert result.applied_edits == [] - assert len(result.messages) == 1 - assert result.messages[0]["role"] == "user" - assert result.messages[0]["content"] == "latest question" + # Post-compaction tail: newer question, newer reply, latest question. + assert [m["role"] for m in result.messages] == ["user", "assistant", "user"] + assert result.messages[0]["content"] == "newer question" + assert result.messages[-1]["content"] == "latest question" + for msg in result.messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + assert block.get("type") != "compaction" def test_client_compaction_block_history_no_compaction_returns_none(): @@ -386,7 +398,13 @@ def test_client_compaction_block_history_no_compaction_returns_none(): async def test_slice_only_path_with_existing_compaction_block(): - """Phase A slices; Phase B token count is below threshold; no summary call.""" + """Phase A slices; Phase B token count is below threshold; no summary call. + + The prior compaction summary lives on the system prefix; the + post-compaction tail is forwarded unchanged so the model retains the + recent turns the summary does not cover. Compaction blocks themselves + are stripped from messages. + """ messages = _messages_with_compaction("prior summary text") with ( @@ -412,10 +430,10 @@ async def test_slice_only_path_with_existing_compaction_block(): assert result.compaction_block is None assert result.iterations_usage is None - # Main call: summary on system, latest user question only (no stale assistant). - assert len(result.messages) == 1 - assert result.messages[0]["role"] == "user" - assert result.messages[0]["content"] == "latest question" + # Main call: summary on system + full post-compaction tail (no compaction blocks). + assert [m["role"] for m in result.messages] == ["user", "assistant", "user"] + assert result.messages[0]["content"] == "newer question" + assert result.messages[-1]["content"] == "latest question" for msg in result.messages: content = msg.get("content") if isinstance(content, list):