diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index 14412403281..1b8237e9812 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -330,14 +330,40 @@ def _extract_summary_text(raw: Optional[str]) -> Optional[str]: return summary or None +def _system_to_openai_message( + system: Optional[Union[str, List[Dict[str, Any]]]], +) -> Optional[Dict[str, Any]]: + """Translate Anthropic-shaped ``system`` to an OpenAI system message. + + Accepts a bare string or a list of Anthropic content blocks; returns + ``None`` if no usable text is present. Only ``type=="text"`` blocks are + carried over — the summary model has no use for ``cache_control`` or + other non-text metadata. + """ + if isinstance(system, str): + return {"role": "system", "content": system} if system else None + if isinstance(system, list): + parts = [ + block.get("text", "") + for block in system + if isinstance(block, dict) and block.get("type") == "text" + ] + joined = "\n\n".join(part for part in parts if part) + return {"role": "system", "content": joined} if joined else None + return None + + def _build_summary_messages( effective_messages: List[Dict[str, Any]], prompt: str, + system: Optional[Union[str, List[Dict[str, Any]]]] = None, ) -> List[Dict[str, Any]]: """Build the OpenAI-shape message list for the summary call. - The conversation history is translated to OpenAI shape; the - summarization prompt is appended as a final user turn. + The caller's ``system`` prompt is prepended (the default summarization + instructions reference "the initial task above", which lives in that + system prompt); the conversation history is translated to OpenAI shape; + the summarization prompt is appended as a final user turn. """ from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( LiteLLMAnthropicMessagesAdapter, @@ -358,7 +384,13 @@ def _build_summary_messages( ) openai_messages = cast(Any, stripped) - return [*openai_messages, {"role": "user", "content": prompt}] + summary_messages: List[Dict[str, Any]] = [] + system_message = _system_to_openai_message(system) + if system_message is not None: + summary_messages.append(system_message) + summary_messages.extend(openai_messages) + summary_messages.append({"role": "user", "content": prompt}) + return summary_messages async def _call_summary_model( @@ -558,7 +590,9 @@ async def apply_compact_20260112( # Phase C: summarize. prompt = _build_summary_prompt(edit_spec, tools) - summary_messages = _build_summary_messages(effective_messages, prompt) + summary_messages = _build_summary_messages( + effective_messages, prompt, system=system + ) propagated_metadata = _propagate_metadata(metadata) try: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index 86d72c91305..e30721c34a9 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -742,6 +742,117 @@ async def test_default_instructions_with_tools_appends_no_tool_suffix(): assert "tool" in prompt.lower() +async def test_system_prompt_forwarded_to_summary_call_as_string(): + """A bare-string ``system`` is prepended as a system message to the summary call.""" + messages = _simple_messages() + mock_response = _make_mock_response("With system") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system="You are a helpful coding agent. The initial task is to fix bug X.", + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + summary_messages = captured_calls[0]["summary_messages"] + assert summary_messages[0]["role"] == "system" + assert "initial task is to fix bug X" in summary_messages[0]["content"] + + +async def test_system_prompt_forwarded_to_summary_call_as_content_blocks(): + """An Anthropic-shaped list ``system`` is flattened to text and prepended.""" + messages = _simple_messages() + mock_response = _make_mock_response("With list system") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + system_blocks = [ + {"type": "text", "text": "Agent role: code reviewer."}, + {"type": "text", "text": "Initial task: review PR #123."}, + ] + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=system_blocks, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + summary_messages = captured_calls[0]["summary_messages"] + assert summary_messages[0]["role"] == "system" + content = summary_messages[0]["content"] + assert "Agent role: code reviewer." in content + assert "Initial task: review PR #123." in content + + +async def test_summary_call_omits_system_message_when_system_is_none(): + """No system message is prepended when the caller did not provide one.""" + messages = _simple_messages() + mock_response = _make_mock_response("No system") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + summary_messages = captured_calls[0]["summary_messages"] + assert all(msg.get("role") != "system" for msg in summary_messages) + + # --------------------------------------------------------------------------- # Dispatcher integration: compact_20260112 via apply_context_management # ---------------------------------------------------------------------------