From 93769dc48d76240e7ded9a95529551c88597aff2 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Thu, 28 May 2026 03:13:53 +0000
Subject: [PATCH] fix(compact): forward caller system prompt to summary model
call
The default summarization instructions reference "the initial task above"
and "the raw history above", but the system prompt that holds that task
was not being forwarded to the summary model. The summary call now
prepends an OpenAI-shaped system message translated from the original
Anthropic-shaped system (str or content-block list) so the summarizer
has the agent role and initial task in scope.
---
.../context_management/editors/compact.py | 42 ++++++-
.../context_management/test_compact.py | 111 ++++++++++++++++++
2 files changed, 149 insertions(+), 4 deletions(-)
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py
index 14412403281..1b8237e9812 100644
--- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py
@@ -330,14 +330,40 @@ def _extract_summary_text(raw: Optional[str]) -> Optional[str]:
return summary or None
+def _system_to_openai_message(
+ system: Optional[Union[str, List[Dict[str, Any]]]],
+) -> Optional[Dict[str, Any]]:
+ """Translate Anthropic-shaped ``system`` to an OpenAI system message.
+
+ Accepts a bare string or a list of Anthropic content blocks; returns
+ ``None`` if no usable text is present. Only ``type=="text"`` blocks are
+ carried over — the summary model has no use for ``cache_control`` or
+ other non-text metadata.
+ """
+ if isinstance(system, str):
+ return {"role": "system", "content": system} if system else None
+ if isinstance(system, list):
+ parts = [
+ block.get("text", "")
+ for block in system
+ if isinstance(block, dict) and block.get("type") == "text"
+ ]
+ joined = "\n\n".join(part for part in parts if part)
+ return {"role": "system", "content": joined} if joined else None
+ return None
+
+
def _build_summary_messages(
effective_messages: List[Dict[str, Any]],
prompt: str,
+ system: Optional[Union[str, List[Dict[str, Any]]]] = None,
) -> List[Dict[str, Any]]:
"""Build the OpenAI-shape message list for the summary call.
- The conversation history is translated to OpenAI shape; the
- summarization prompt is appended as a final user turn.
+ The caller's ``system`` prompt is prepended (the default summarization
+ instructions reference "the initial task above", which lives in that
+ system prompt); the conversation history is translated to OpenAI shape;
+ the summarization prompt is appended as a final user turn.
"""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
@@ -358,7 +384,13 @@ def _build_summary_messages(
)
openai_messages = cast(Any, stripped)
- return [*openai_messages, {"role": "user", "content": prompt}]
+ summary_messages: List[Dict[str, Any]] = []
+ system_message = _system_to_openai_message(system)
+ if system_message is not None:
+ summary_messages.append(system_message)
+ summary_messages.extend(openai_messages)
+ summary_messages.append({"role": "user", "content": prompt})
+ return summary_messages
async def _call_summary_model(
@@ -558,7 +590,9 @@ async def apply_compact_20260112(
# Phase C: summarize.
prompt = _build_summary_prompt(edit_spec, tools)
- summary_messages = _build_summary_messages(effective_messages, prompt)
+ summary_messages = _build_summary_messages(
+ effective_messages, prompt, system=system
+ )
propagated_metadata = _propagate_metadata(metadata)
try:
diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py
index 86d72c91305..e30721c34a9 100644
--- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py
+++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py
@@ -742,6 +742,117 @@ async def test_default_instructions_with_tools_appends_no_tool_suffix():
assert "tool" in prompt.lower()
+async def test_system_prompt_forwarded_to_summary_call_as_string():
+ """A bare-string ``system`` is prepended as a system message to the summary call."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("With system")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system="You are a helpful coding agent. The initial task is to fix bug X.",
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ summary_messages = captured_calls[0]["summary_messages"]
+ assert summary_messages[0]["role"] == "system"
+ assert "initial task is to fix bug X" in summary_messages[0]["content"]
+
+
+async def test_system_prompt_forwarded_to_summary_call_as_content_blocks():
+ """An Anthropic-shaped list ``system`` is flattened to text and prepended."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("With list system")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ system_blocks = [
+ {"type": "text", "text": "Agent role: code reviewer."},
+ {"type": "text", "text": "Initial task: review PR #123."},
+ ]
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=system_blocks,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ summary_messages = captured_calls[0]["summary_messages"]
+ assert summary_messages[0]["role"] == "system"
+ content = summary_messages[0]["content"]
+ assert "Agent role: code reviewer." in content
+ assert "Initial task: review PR #123." in content
+
+
+async def test_summary_call_omits_system_message_when_system_is_none():
+ """No system message is prepended when the caller did not provide one."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("No system")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ summary_messages = captured_calls[0]["summary_messages"]
+ assert all(msg.get("role") != "system" for msg in summary_messages)
+
+
# ---------------------------------------------------------------------------
# Dispatcher integration: compact_20260112 via apply_context_management
# ---------------------------------------------------------------------------