fix(compact): forward caller system prompt to summary model call

The default summarization instructions reference "the initial task above"
and "the raw history above", but the system prompt that holds that task
was not being forwarded to the summary model. The summary call now
prepends an OpenAI-shaped system message translated from the original
Anthropic-shaped system (str or content-block list) so the summarizer
has the agent role and initial task in scope.
This commit is contained in:
mateo-berri 2026-05-28 03:13:53 +00:00
parent b17a8c2acf
commit 93769dc48d
No known key found for this signature in database
2 changed files with 149 additions and 4 deletions

View file

@ -330,14 +330,40 @@ def _extract_summary_text(raw: Optional[str]) -> Optional[str]:
return summary or None
def _system_to_openai_message(
system: Optional[Union[str, List[Dict[str, Any]]]],
) -> Optional[Dict[str, Any]]:
"""Translate Anthropic-shaped ``system`` to an OpenAI system message.
Accepts a bare string or a list of Anthropic content blocks; returns
``None`` if no usable text is present. Only ``type=="text"`` blocks are
carried over — the summary model has no use for ``cache_control`` or
other non-text metadata.
"""
if isinstance(system, str):
return {"role": "system", "content": system} if system else None
if isinstance(system, list):
parts = [
block.get("text", "")
for block in system
if isinstance(block, dict) and block.get("type") == "text"
]
joined = "\n\n".join(part for part in parts if part)
return {"role": "system", "content": joined} if joined else None
return None
def _build_summary_messages(
effective_messages: List[Dict[str, Any]],
prompt: str,
system: Optional[Union[str, List[Dict[str, Any]]]] = None,
) -> List[Dict[str, Any]]:
"""Build the OpenAI-shape message list for the summary call.
The conversation history is translated to OpenAI shape; the
summarization prompt is appended as a final user turn.
The caller's ``system`` prompt is prepended (the default summarization
instructions reference "the initial task above", which lives in that
system prompt); the conversation history is translated to OpenAI shape;
the summarization prompt is appended as a final user turn.
"""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
@ -358,7 +384,13 @@ def _build_summary_messages(
)
openai_messages = cast(Any, stripped)
return [*openai_messages, {"role": "user", "content": prompt}]
summary_messages: List[Dict[str, Any]] = []
system_message = _system_to_openai_message(system)
if system_message is not None:
summary_messages.append(system_message)
summary_messages.extend(openai_messages)
summary_messages.append({"role": "user", "content": prompt})
return summary_messages
async def _call_summary_model(
@ -558,7 +590,9 @@ async def apply_compact_20260112(
# Phase C: summarize.
prompt = _build_summary_prompt(edit_spec, tools)
summary_messages = _build_summary_messages(effective_messages, prompt)
summary_messages = _build_summary_messages(
effective_messages, prompt, system=system
)
propagated_metadata = _propagate_metadata(metadata)
try:

View file

@ -742,6 +742,117 @@ async def test_default_instructions_with_tools_appends_no_tool_suffix():
assert "tool" in prompt.lower()
async def test_system_prompt_forwarded_to_summary_call_as_string():
"""A bare-string ``system`` is prepended as a system message to the summary call."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>With system</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system="You are a helpful coding agent. The initial task is to fix bug X.",
edit_spec=_EDIT_SPEC_DEFAULT,
)
summary_messages = captured_calls[0]["summary_messages"]
assert summary_messages[0]["role"] == "system"
assert "initial task is to fix bug X" in summary_messages[0]["content"]
async def test_system_prompt_forwarded_to_summary_call_as_content_blocks():
"""An Anthropic-shaped list ``system`` is flattened to text and prepended."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>With list system</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
system_blocks = [
{"type": "text", "text": "Agent role: code reviewer."},
{"type": "text", "text": "Initial task: review PR #123."},
]
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=system_blocks,
edit_spec=_EDIT_SPEC_DEFAULT,
)
summary_messages = captured_calls[0]["summary_messages"]
assert summary_messages[0]["role"] == "system"
content = summary_messages[0]["content"]
assert "Agent role: code reviewer." in content
assert "Initial task: review PR #123." in content
async def test_summary_call_omits_system_message_when_system_is_none():
"""No system message is prepended when the caller did not provide one."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>No system</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
summary_messages = captured_calls[0]["summary_messages"]
assert all(msg.get("role") != "system" for msg in summary_messages)
# ---------------------------------------------------------------------------
# Dispatcher integration: compact_20260112 via apply_context_management
# ---------------------------------------------------------------------------