From 3926fd09b641fec83c999decdc624d585d638046 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 28 May 2026 03:16:56 +0000 Subject: [PATCH] fix(compact_20260112): set default max_tokens and merge prompt when last turn is user - Set COMPACT_SUMMARY_MAX_TOKENS default for the summary call so providers like Anthropic (which require max_tokens) don't silently fail and degrade to summary_call_failed. - When the trailing translated message is already a user turn, merge the summarization prompt into it instead of appending a second user turn. Avoids consecutive role=user messages that strict providers reject. Co-authored-by: Yassin Kortam --- .../context_management/constants.py | 5 ++ .../context_management/editors/compact.py | 38 ++++++++- .../context_management/test_compact.py | 84 ++++++++++++++++++- 3 files changed, 123 insertions(+), 4 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py index 0d165698f23..605e8cb6b99 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py @@ -11,6 +11,11 @@ CLEARED_TOOL_RESULT_PLACEHOLDER = "[Cleared by context management]" COMPACT_EDIT_TYPE = "compact_20260112" COMPACT_DEFAULT_TRIGGER_TOKENS = 150_000 COMPACT_MIN_TRIGGER_TOKENS = 50_000 +# Default ``max_tokens`` for the summary call. Required by providers like +# Anthropic that reject requests without it; safely accepted by providers that +# don't strictly require it. Chosen to comfortably fit a long structured +# summary. +COMPACT_SUMMARY_MAX_TOKENS = 4096 COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model" COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: " diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index 1b8237e9812..3ebb4a2e6e4 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -29,6 +29,7 @@ from ..constants import ( COMPACT_EDIT_TYPE, COMPACT_MIN_TRIGGER_TOKENS, COMPACT_NO_TOOL_CALLS_SUFFIX, + COMPACT_SUMMARY_MAX_TOKENS, COMPACT_SUMMARY_MODEL_SETTING_KEY, COMPACT_SUMMARY_SYSTEM_PREFIX, ) @@ -389,10 +390,40 @@ def _build_summary_messages( if system_message is not None: summary_messages.append(system_message) summary_messages.extend(openai_messages) - summary_messages.append({"role": "user", "content": prompt}) + # If the last turn is already a user message, merge the summarization + # prompt into it. Some providers (and strict OpenAI-compatible endpoints) + # reject two consecutive ``role=user`` messages, which would otherwise + # silently fall into the ``summary_call_failed`` error path. + if summary_messages and _is_user_message(summary_messages[-1]): + last_msg = summary_messages[-1] + summary_messages[-1] = { + **last_msg, + "content": _append_text_to_content(last_msg.get("content"), prompt), + } + else: + summary_messages.append({"role": "user", "content": prompt}) return summary_messages +def _is_user_message(msg: Any) -> bool: + return isinstance(msg, dict) and msg.get("role") == "user" + + +def _append_text_to_content(content: Any, extra_text: str) -> Any: + """Append ``extra_text`` to an OpenAI-shape message ``content`` field. + + Handles the two common shapes: ``str`` and ``list`` of content parts. + For unexpected/empty shapes, fall back so the caller gets a usable value. + """ + if content is None or content == "": + return extra_text + if isinstance(content, str): + return f"{content}\n\n{extra_text}" + if isinstance(content, list): + return [*content, {"type": "text", "text": extra_text}] + return [content, {"type": "text", "text": extra_text}] + + async def _call_summary_model( *, summary_model: str, @@ -406,9 +437,14 @@ async def _call_summary_model( proxy's ``model_list``; falls back to ``litellm.acompletion`` if no router is available (e.g. SDK usage outside the proxy). """ + # ``max_tokens`` is required by providers like Anthropic and silently + # accepted by providers that don't strictly require it (OpenAI etc.). + # Setting a sensible default here means the feature works regardless of + # which model an admin configures as ``context_management_summary_model``. call_kwargs: Dict[str, Any] = { "model": summary_model, "messages": summary_messages, + "max_tokens": COMPACT_SUMMARY_MAX_TOKENS, "metadata": metadata, } if llm_router is not None and hasattr(llm_router, "acompletion"): diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index e30721c34a9..8944b76979e 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -664,12 +664,14 @@ async def test_custom_instructions_used_verbatim(): assert len(captured_calls) == 1 summary_messages = captured_calls[0]["summary_messages"] - # The last message should be the custom instruction prompt + # The custom instruction prompt is appended to the trailing user turn so + # we don't end up with two consecutive ``role=user`` messages (some + # providers reject that). last_msg = summary_messages[-1] assert last_msg["role"] == "user" - assert last_msg["content"] == "Summarize everything briefly." + assert "Summarize everything briefly." in last_msg["content"] # The "do not call tools" suffix should NOT be in the prompt since custom was set - assert "tool" not in last_msg["content"].lower() + assert "do not call" not in last_msg["content"].lower() async def test_default_instructions_appended_with_no_tool_suffix_when_no_tools(): @@ -853,6 +855,82 @@ async def test_summary_call_omits_system_message_when_system_is_none(): assert all(msg.get("role") != "system" for msg in summary_messages) +async def test_summary_call_does_not_emit_consecutive_user_turns(): + """When the trailing message is already a user turn, the summarization + prompt is merged into it instead of appended as a second user message. + + Some providers (and strict OpenAI-compatible endpoints) reject two + consecutive ``role=user`` messages, which would silently fall into the + ``summary_call_failed`` error path. + """ + messages = _simple_messages() + assert messages[-1]["role"] == "user" + mock_response = _make_mock_response("x") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + summary_messages = captured_calls[0]["summary_messages"] + user_indices = [ + idx for idx, msg in enumerate(summary_messages) if msg.get("role") == "user" + ] + # No two adjacent indices. + assert all( + b - a > 1 for a, b in zip(user_indices, user_indices[1:]) + ), f"two consecutive user turns produced: {summary_messages}" + + +async def test_summary_call_sends_default_max_tokens(): + """``max_tokens`` is set on the summary call so providers like Anthropic + (which require it) don't reject the request and silently fall back to + ``summary_call_failed``. + """ + from litellm.llms.anthropic.experimental_pass_through.context_management.constants import ( + COMPACT_SUMMARY_MAX_TOKENS, + ) + from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _call_summary_model, + ) + + captured_kwargs: dict = {} + + class _FakeRouter: + async def acompletion(self, **kwargs): + captured_kwargs.update(kwargs) + return _make_mock_response("x") + + await _call_summary_model( + summary_model="claude-haiku-4-5", + summary_messages=[{"role": "user", "content": "hi"}], + metadata={}, + llm_router=_FakeRouter(), + ) + + assert captured_kwargs.get("max_tokens") == COMPACT_SUMMARY_MAX_TOKENS + + # --------------------------------------------------------------------------- # Dispatcher integration: compact_20260112 via apply_context_management # ---------------------------------------------------------------------------