diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
index 6908921a43f..7afe3334f29 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py
@@ -12,13 +12,18 @@ from typing import (
)
import litellm
+from litellm._logging import verbose_logger
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
AnthropicAdapter,
)
+from litellm.llms.anthropic.experimental_pass_through.context_management import (
+ AnthropicContextManagementError,
+ PolyfillResult,
+ apply_context_management,
+)
from litellm.llms.anthropic.experimental_pass_through.utils import (
is_reasoning_auto_summary_enabled,
)
-from litellm.types.llms.anthropic import AppliedEdit
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
@@ -31,9 +36,58 @@ if TYPE_CHECKING:
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset(
- {"output_config", "context_management", "_polyfill_applied_edits"}
+ {"output_config", "context_management", "_polyfill_result"}
)
+
+async def _run_polyfill_if_enabled(
+ *,
+ model: str,
+ messages: List[Dict],
+ tools: Optional[List[Dict]],
+ system: Optional[Any],
+ context_management_spec: Any,
+ metadata: Optional[Dict],
+ drop_params: Optional[bool],
+ llm_router: Any,
+) -> Optional[PolyfillResult]:
+ """Run the async context_management polyfill if a spec is present.
+
+ Returns ``None`` when the spec is empty or drop_params is on. Raises
+ ``AnthropicContextManagementError`` so the /v1/messages endpoint can
+ emit an Anthropic-format 400. All other exceptions are best-effort
+ swallowed (matches v0 behavior).
+ """
+ if not context_management_spec:
+ return None
+
+ effective_drop_params = (
+ drop_params if drop_params is not None else litellm.drop_params
+ )
+ if effective_drop_params:
+ return None
+
+ try:
+ return await apply_context_management(
+ model=model,
+ messages=messages,
+ tools=tools,
+ system=system,
+ context_management_spec=context_management_spec,
+ metadata=metadata,
+ llm_router=llm_router,
+ )
+ except AnthropicContextManagementError:
+ # Surface validation errors so the endpoint can emit an Anthropic-format
+ # 400. Other exception types fall into the best-effort branch below.
+ raise
+ except Exception as e:
+ verbose_logger.exception(
+ "context_management polyfill: skipping edits due to error: %s", e
+ )
+ return None
+
+
########################################################
# init adapter
ANTHROPIC_ADAPTER = AnthropicAdapter()
@@ -303,21 +357,49 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_k: Optional[int] = None,
top_p: Optional[float] = None,
output_format: Optional[Dict] = None,
- _polyfill_applied_edits: Optional[List[AppliedEdit]] = None,
**kwargs,
) -> Union[AnthropicMessagesResponse, AsyncIterator]:
"""Handle non-Anthropic models asynchronously using the adapter"""
+ context_management = kwargs.pop("context_management", None)
+ drop_params: Optional[bool] = kwargs.get("drop_params", None)
+ litellm_router = kwargs.pop("litellm_router", None)
+ if litellm_router is None:
+ try:
+ from litellm.proxy.proxy_server import llm_router as _proxy_router
+
+ litellm_router = _proxy_router
+ except Exception:
+ pass
+
+ polyfill_result = await _run_polyfill_if_enabled(
+ model=model,
+ messages=messages,
+ tools=tools,
+ system=system,
+ context_management_spec=context_management,
+ metadata=metadata,
+ drop_params=drop_params,
+ llm_router=litellm_router,
+ )
+
+ effective_messages = (
+ polyfill_result.messages if polyfill_result is not None else messages
+ )
+ effective_system = (
+ polyfill_result.system if polyfill_result is not None else system
+ )
+
(
completion_kwargs,
tool_name_mapping,
) = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
max_tokens=max_tokens,
- messages=messages,
+ messages=effective_messages,
model=model,
metadata=metadata,
stop_sequences=stop_sequences,
stream=stream,
- system=system,
+ system=effective_system,
temperature=temperature,
thinking=thinking,
tool_choice=tool_choice,
@@ -336,7 +418,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
completion_response,
model=model,
tool_name_mapping=tool_name_mapping,
- applied_edits=_polyfill_applied_edits,
+ polyfill_result=polyfill_result,
)
)
if transformed_stream is not None:
@@ -346,7 +428,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params(
cast(ModelResponse, completion_response),
tool_name_mapping=tool_name_mapping,
- applied_edits=_polyfill_applied_edits,
+ polyfill_result=polyfill_result,
)
if anthropic_response is not None:
return anthropic_response
@@ -369,7 +451,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_p: Optional[float] = None,
output_format: Optional[Dict] = None,
_is_async: bool = False,
- _polyfill_applied_edits: Optional[List[AppliedEdit]] = None,
**kwargs,
) -> Union[
AnthropicMessagesResponse,
@@ -393,7 +474,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_k=top_k,
top_p=top_p,
output_format=output_format,
- _polyfill_applied_edits=_polyfill_applied_edits,
**kwargs,
)
@@ -426,7 +506,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
completion_response,
model=model,
tool_name_mapping=tool_name_mapping,
- applied_edits=_polyfill_applied_edits,
)
)
if transformed_stream is not None:
@@ -436,7 +515,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params(
cast(ModelResponse, completion_response),
tool_name_mapping=tool_name_mapping,
- applied_edits=_polyfill_applied_edits,
)
if anthropic_response is not None:
return anthropic_response
diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
index 9fc88874b13..4069bd7fd31 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
@@ -75,6 +75,9 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
from litellm.litellm_core_utils.prompt_templates.factory import (
THOUGHT_SIGNATURE_SEPARATOR,
)
+from litellm.llms.anthropic.experimental_pass_through.context_management import (
+ PolyfillResult,
+)
from litellm.types.llms.anthropic import (
ANTHROPIC_HOSTED_TOOLS,
AllAnthropicToolsValues,
@@ -88,6 +91,7 @@ from litellm.types.llms.anthropic import (
AnthropicResponseContentBlockThinking,
AnthropicResponseContentBlockToolUse,
AppliedEdit,
+ CompactionBlock,
ContentBlockDelta,
ContentJsonBlockDelta,
ContentTextBlockDelta,
@@ -97,6 +101,7 @@ from litellm.types.llms.anthropic import (
MessageBlockDelta,
MessageDelta,
UsageDelta,
+ UsageIteration,
)
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
@@ -197,7 +202,7 @@ class AnthropicAdapter:
self,
response: ModelResponse,
tool_name_mapping: Optional[Dict[str, str]] = None,
- applied_edits: Optional[List[AppliedEdit]] = None,
+ polyfill_result: Optional[PolyfillResult] = None,
) -> Optional[AnthropicMessagesResponse]:
"""
Translate OpenAI response to Anthropic format.
@@ -207,12 +212,12 @@ class AnthropicAdapter:
tool_name_mapping: Optional mapping of truncated tool names to original names.
Used to restore original names for tools that exceeded
OpenAI's 64-char limit.
- applied_edits: Polyfill AppliedEdit list for response context_management.
+ polyfill_result: PolyfillResult from context_management polyfill.
"""
return LiteLLMAnthropicMessagesAdapter().translate_openai_response_to_anthropic(
response=response,
tool_name_mapping=tool_name_mapping,
- applied_edits=applied_edits,
+ polyfill_result=polyfill_result,
)
def translate_completion_output_params_streaming(
@@ -220,7 +225,7 @@ class AnthropicAdapter:
completion_stream: Any,
model: str,
tool_name_mapping: Optional[Dict[str, str]] = None,
- applied_edits: Optional[List[AppliedEdit]] = None,
+ polyfill_result: Optional[PolyfillResult] = None,
) -> Union[AsyncIterator[bytes], None]:
"""
Translate OpenAI streaming response to Anthropic format.
@@ -229,8 +234,9 @@ class AnthropicAdapter:
completion_stream: The OpenAI streaming response
model: The model name
tool_name_mapping: Optional mapping of truncated tool names to original names.
- applied_edits: Polyfill AppliedEdit list on final message_delta.
+ polyfill_result: PolyfillResult from context_management polyfill.
"""
+ applied_edits = polyfill_result.applied_edits if polyfill_result else None
anthropic_wrapper = AnthropicStreamWrapper(
completion_stream=completion_stream,
model=model,
@@ -1350,7 +1356,7 @@ class LiteLLMAnthropicMessagesAdapter:
self,
response: ModelResponse,
tool_name_mapping: Optional[Dict[str, str]] = None,
- applied_edits: Optional[List[AppliedEdit]] = None,
+ polyfill_result: Optional[PolyfillResult] = None,
) -> AnthropicMessagesResponse:
"""
Translate OpenAI response to Anthropic format.
@@ -1360,13 +1366,17 @@ class LiteLLMAnthropicMessagesAdapter:
tool_name_mapping: Optional mapping of truncated tool names to original names.
Used to restore original names for tools that exceeded
OpenAI's 64-char limit.
- applied_edits: Polyfill AppliedEdit list for response context_management.
+ polyfill_result: PolyfillResult from context_management polyfill.
"""
## translate content block
anthropic_content = self._translate_openai_content_to_anthropic(
choices=response.choices, # type: ignore
tool_name_mapping=tool_name_mapping,
)
+
+ if polyfill_result is not None and polyfill_result.compaction_block is not None:
+ anthropic_content.insert(0, polyfill_result.compaction_block) # type: ignore[arg-type]
+
## extract finish reason
anthropic_finish_reason = self._translate_openai_finish_reason_to_anthropic(
openai_finish_reason=response.choices[0].finish_reason # type: ignore
@@ -1395,6 +1405,14 @@ class LiteLLMAnthropicMessagesAdapter:
if cached_tokens > 0:
anthropic_usage["cache_read_input_tokens"] = cached_tokens
+ if polyfill_result is not None and polyfill_result.iterations_usage is not None:
+ message_iteration: UsageIteration = {
+ "type": "message",
+ "input_tokens": uncached_input_tokens,
+ "output_tokens": usage.completion_tokens or 0,
+ }
+ anthropic_usage["iterations"] = list(polyfill_result.iterations_usage) + [message_iteration] # type: ignore[typeddict-unknown-key]
+
translated_obj = AnthropicMessagesResponse(
id=response.id,
type="message",
@@ -1406,6 +1424,7 @@ class LiteLLMAnthropicMessagesAdapter:
stop_reason=anthropic_finish_reason,
)
+ applied_edits = polyfill_result.applied_edits if polyfill_result else None
if applied_edits:
translated_obj["context_management"] = ContextManagementResponse(
applied_edits=list(applied_edits)
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py b/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py
index f9bd9f4c66d..729b2864524 100644
--- a/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py
@@ -1,4 +1,11 @@
from .constants import CLEARED_TOOL_RESULT_PLACEHOLDER
from .dispatcher import apply_context_management
+from .errors import AnthropicContextManagementError
+from .result import PolyfillResult
-__all__ = ["apply_context_management", "CLEARED_TOOL_RESULT_PLACEHOLDER"]
+__all__ = [
+ "apply_context_management",
+ "AnthropicContextManagementError",
+ "CLEARED_TOOL_RESULT_PLACEHOLDER",
+ "PolyfillResult",
+]
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py
index 4003591c0d4..0d165698f23 100644
--- a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py
@@ -6,3 +6,28 @@ DEFAULT_INPUT_TOKENS_TRIGGER = 100_000
DEFAULT_KEEP_TOOL_USES = 3
CLEARED_TOOL_RESULT_PLACEHOLDER = "[Cleared by context management]"
+
+# compact_20260112
+COMPACT_EDIT_TYPE = "compact_20260112"
+COMPACT_DEFAULT_TRIGGER_TOKENS = 150_000
+COMPACT_MIN_TRIGGER_TOKENS = 50_000
+COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model"
+COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: "
+
+# Default summarization prompt from the Anthropic spec.
+COMPACT_DEFAULT_INSTRUCTIONS = (
+ "You have written a partial transcript for the initial task above. Please "
+ "write a summary of the transcript. The purpose of this summary is to "
+ "provide continuity so you can continue to make progress towards solving "
+ "the task in a future context, where the raw history above may not be "
+ "accessible and will be replaced with this summary. Write down anything "
+ "that would be helpful, including the state, next steps, learnings etc. "
+ "You must wrap your summary in a block."
+)
+
+# Appended to the default prompt when ``tools`` are present and the caller
+# did not supply custom ``instructions``. Matches the guidance in the
+# Anthropic docs under "Compaction might fail when tools are defined".
+COMPACT_NO_TOOL_CALLS_SUFFIX = (
+ " Do not call any tools while writing this summary; respond with text only."
+)
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py b/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py
index 93881580e92..ab3a905c64c 100644
--- a/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py
@@ -1,56 +1,85 @@
"""Dispatch ``context_management`` edits to registered polyfill editors."""
-from typing import Any, Callable, Dict, List, Optional, Tuple, Union
+import inspect
+from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple, Union, cast
from litellm._logging import verbose_logger
-from litellm.types.llms.anthropic import AppliedEdit
-from .constants import CLEAR_TOOL_USES_EDIT_TYPE
-from .editors import apply_clear_tool_uses_20250919
+from .constants import CLEAR_TOOL_USES_EDIT_TYPE, COMPACT_EDIT_TYPE
+from .editors import apply_clear_tool_uses_20250919, apply_compact_20260112
+from .result import PolyfillResult
-EditorFn = Callable[..., Tuple[List[Dict[str, Any]], Optional[AppliedEdit]]]
+EditorFn = Callable[..., Any]
_EDITOR_REGISTRY: Dict[str, EditorFn] = {
CLEAR_TOOL_USES_EDIT_TYPE: apply_clear_tool_uses_20250919,
+ COMPACT_EDIT_TYPE: apply_compact_20260112,
}
-def apply_context_management(
+def _normalize_spec(
+ spec: Union[Dict[str, Any], List[Dict[str, Any]], None],
+) -> Optional[List[Dict[str, Any]]]:
+ """Accept Anthropic-native dict form or OpenAI list form; return edits list."""
+ if isinstance(spec, list):
+ # Local import to avoid an import cycle at module load.
+ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
+
+ spec = AnthropicConfig.map_openai_context_management_to_anthropic(spec)
+
+ edits = spec.get("edits") if isinstance(spec, dict) else None
+ if not edits or not isinstance(edits, list):
+ return None
+ return [edit for edit in edits if isinstance(edit, dict)]
+
+
+def _wrap_editor_return(raw: Any, *, fallback_system: Any) -> PolyfillResult:
+ """Coerce an editor's native return shape into a ``PolyfillResult``.
+
+ v0 sync editors (e.g. ``clear_tool_uses_20250919``) return a 2-tuple
+ ``(messages, Optional[AppliedEdit])``. The new async ``compact_20260112``
+ editor returns a ``PolyfillResult`` directly.
+ """
+ if isinstance(raw, PolyfillResult):
+ return raw
+ # Legacy 2-tuple return — sync editors don't mutate ``system``, so
+ # carry the caller's value forward.
+ messages, applied = cast(Tuple[List[Dict[str, Any]], Any], raw)
+ return PolyfillResult(
+ messages=messages,
+ system=fallback_system,
+ applied_edits=[applied] if applied is not None else [],
+ )
+
+
+async def apply_context_management(
*,
model: str,
messages: List[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]],
system: Any,
context_management_spec: Union[Dict[str, Any], List[Dict[str, Any]], None],
-) -> Tuple[List[Dict[str, Any]], List[AppliedEdit]]:
- """Run edits in order; return (messages, applied_edits that fired)."""
- # Accept both Anthropic-native dict form and OpenAI list form. The other
- # provider paths normalize via ``map_openai_context_management_to_anthropic``
- # before dispatching; do the same here so the polyfill path doesn't silently
- # no-op on list input.
- if isinstance(context_management_spec, list):
- from litellm.llms.anthropic.chat.transformation import AnthropicConfig
+ metadata: Optional[Dict[str, Any]] = None,
+ llm_router: Any = None,
+) -> PolyfillResult:
+ """Run edits in order; return a single ``PolyfillResult``.
- context_management_spec = (
- AnthropicConfig.map_openai_context_management_to_anthropic(
- context_management_spec
- )
- )
+ The dispatcher is async so async editors (``compact_20260112``) can
+ ``await`` the configured summarization model. Sync editors are called
+ inline — ``inspect.iscoroutinefunction`` decides how each editor is
+ invoked.
+ """
+ edits = _normalize_spec(context_management_spec)
+ if not edits:
+ return PolyfillResult(messages=messages, system=system, applied_edits=[])
- edits = (
- context_management_spec.get("edits")
- if isinstance(context_management_spec, dict)
- else None
- )
- if not edits or not isinstance(edits, list):
- return messages, []
-
- applied_edits: List[AppliedEdit] = []
current_messages = messages
+ current_system = system
+ aggregated_applied: List[Dict[str, Any]] = []
+ aggregated_compaction_block = None
+ aggregated_iterations_usage = None
for edit_spec in edits:
- if not isinstance(edit_spec, dict):
- continue
edit_type = edit_spec.get("type")
editor = _EDITOR_REGISTRY.get(edit_type) if isinstance(edit_type, str) else None
if editor is None:
@@ -60,14 +89,36 @@ def apply_context_management(
)
continue
- current_messages, applied = editor(
- model=model,
- messages=current_messages,
- tools=tools,
- system=system,
- edit_spec=edit_spec,
- )
- if applied is not None:
- applied_edits.append(applied)
+ kwargs: Dict[str, Any] = {
+ "model": model,
+ "messages": current_messages,
+ "tools": tools,
+ "system": current_system,
+ "edit_spec": edit_spec,
+ }
+ # Only async editors accept these — passing them to sync v0 editors
+ # would break their signature.
+ if inspect.iscoroutinefunction(editor):
+ kwargs["metadata"] = metadata
+ kwargs["llm_router"] = llm_router
+ raw_result = await cast(Callable[..., Awaitable[Any]], editor)(**kwargs)
+ else:
+ raw_result = editor(**kwargs)
- return current_messages, applied_edits
+ result = _wrap_editor_return(raw_result, fallback_system=current_system)
+
+ current_messages = result.messages
+ current_system = result.system
+ aggregated_applied.extend(result.applied_edits)
+ if result.compaction_block is not None:
+ aggregated_compaction_block = result.compaction_block
+ if result.iterations_usage is not None:
+ aggregated_iterations_usage = result.iterations_usage
+
+ return PolyfillResult(
+ messages=current_messages,
+ system=current_system,
+ applied_edits=cast(List[Any], aggregated_applied),
+ compaction_block=aggregated_compaction_block,
+ iterations_usage=aggregated_iterations_usage,
+ )
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py
index d402632380d..3e933a9880a 100644
--- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py
@@ -1,3 +1,4 @@
from .clear_tool_uses import apply_clear_tool_uses_20250919
+from .compact import apply_compact_20260112
-__all__ = ["apply_clear_tool_uses_20250919"]
+__all__ = ["apply_clear_tool_uses_20250919", "apply_compact_20260112"]
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py
new file mode 100644
index 00000000000..62e4e2c3602
--- /dev/null
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py
@@ -0,0 +1,494 @@
+"""``compact_20260112`` polyfill (server-side context compaction).
+
+Mirrors Anthropic's native ``compact_20260112`` for non-Anthropic providers:
+
+- Scans the message history for an existing ``compaction`` block; everything
+ before it is dropped (slice).
+- If still over the configured trigger, calls a separately-configured
+ summarization model and synthesizes a fresh ``compaction`` block.
+- The summary is injected as a system-message prefix on the downstream call
+ (the user/assistant log carries no ``compaction`` block downstream).
+- The synthesized ``compaction`` block is returned via ``PolyfillResult`` so
+ the response adapter can prepend it to the response ``content`` array.
+"""
+
+import re
+from typing import Any, Dict, List, Optional, Tuple, Union, cast
+
+import litellm
+from litellm._logging import verbose_logger
+from litellm.types.llms.anthropic import (
+ AppliedEdit,
+ CompactionBlock,
+ UsageIteration,
+)
+
+from ..constants import (
+ COMPACT_DEFAULT_INSTRUCTIONS,
+ COMPACT_DEFAULT_TRIGGER_TOKENS,
+ COMPACT_EDIT_TYPE,
+ COMPACT_MIN_TRIGGER_TOKENS,
+ COMPACT_NO_TOOL_CALLS_SUFFIX,
+ COMPACT_SUMMARY_MODEL_SETTING_KEY,
+ COMPACT_SUMMARY_SYSTEM_PREFIX,
+)
+from ..errors import AnthropicContextManagementError
+from ..result import PolyfillResult
+
+# Auth metadata fields propagated from the parent request to the summary call
+# so the summary's spend is attributed to the same team/key. The list mirrors
+# the fields populated by
+# ``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata``.
+_PROPAGATED_METADATA_KEYS = (
+ "user_api_key",
+ "user_api_key_alias",
+ "user_api_key_team_id",
+ "user_api_key_team_alias",
+ "user_api_key_user_id",
+ "user_api_key_user_email",
+ "user_api_key_org_id",
+ "litellm_call_id",
+ "litellm_parent_otel_span",
+)
+
+_SUMMARY_TAG_RE = re.compile(r"(.*?)", re.IGNORECASE | re.DOTALL)
+
+
+def _read_summary_model_setting() -> Optional[str]:
+ """Look up the configured summarization model from proxy general_settings."""
+ try:
+ from litellm.proxy.proxy_server import general_settings
+ except Exception:
+ return None
+ value = general_settings.get(COMPACT_SUMMARY_MODEL_SETTING_KEY)
+ return value if isinstance(value, str) and value else None
+
+
+def _find_latest_compaction_index(
+ messages: List[Dict[str, Any]],
+) -> Tuple[Optional[int], Optional[int]]:
+ """Return (message_index, block_index) of the most recent compaction block.
+
+ ``None, None`` if no compaction block is present. Iterates from the end so
+ only the latest one is considered.
+ """
+ for msg_idx in range(len(messages) - 1, -1, -1):
+ content = messages[msg_idx].get("content")
+ if not isinstance(content, list):
+ continue
+ for blk_idx in range(len(content) - 1, -1, -1):
+ block = content[blk_idx]
+ if isinstance(block, dict) and block.get("type") == "compaction":
+ return msg_idx, blk_idx
+ return None, None
+
+
+def _slice_around_compaction_block(
+ messages: List[Dict[str, Any]],
+) -> Tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]]:
+ """Apply Anthropic's "drop everything before the compaction block" rule.
+
+ Returns ``(sliced_messages_with_compaction_block, compaction_block_dict)``
+ if a block was found, else ``(original_messages, None)``. The sliced result
+ keeps the compaction block in the assistant turn that originally carried
+ it (in practice it's the only block in that turn) so callers can still
+ extract the summary text from it.
+ """
+ msg_idx, blk_idx = _find_latest_compaction_index(messages)
+ if msg_idx is None or blk_idx is None:
+ return messages, None
+
+ original_msg = messages[msg_idx]
+ original_content = original_msg["content"]
+ compaction_block = cast(Dict[str, Any], original_content[blk_idx])
+
+ # Per Anthropic's contract everything before the compaction block is
+ # dropped, including earlier blocks within the same assistant message.
+ sliced_content = list(original_content[blk_idx:])
+ sliced_first_msg = {**original_msg, "content": sliced_content}
+
+ sliced_messages: List[Dict[str, Any]] = [sliced_first_msg]
+ sliced_messages.extend(messages[msg_idx + 1 :])
+ return sliced_messages, compaction_block
+
+
+def _strip_compaction_blocks(
+ messages: List[Dict[str, Any]],
+) -> List[Dict[str, Any]]:
+ """Drop any ``compaction`` content blocks from messages.
+
+ Used to build the downstream-bound message list — the adapter has no
+ concept of a compaction block, so it must not see one.
+ """
+ cleaned: List[Dict[str, Any]] = []
+ for msg in messages:
+ content = msg.get("content")
+ if not isinstance(content, list):
+ cleaned.append(msg)
+ continue
+ filtered = [
+ block
+ for block in content
+ if not (isinstance(block, dict) and block.get("type") == "compaction")
+ ]
+ if not filtered:
+ # The compaction block was the only content; drop the whole turn.
+ continue
+ cleaned.append({**msg, "content": filtered})
+ return cleaned
+
+
+def _augment_system_with_summary(
+ system: Optional[Union[str, List[Dict[str, Any]]]],
+ summary_text: str,
+) -> Union[str, List[Dict[str, Any]]]:
+ """Prepend a "Previous conversation summary: ..." block to ``system``."""
+ prefix = f"{COMPACT_SUMMARY_SYSTEM_PREFIX}{summary_text}\n\n"
+ if system is None:
+ return prefix.rstrip()
+ if isinstance(system, str):
+ return f"{prefix}{system}"
+ # List of content blocks: prepend the prefix to the first text block,
+ # otherwise insert a new text block at the head.
+ for idx, block in enumerate(system):
+ if isinstance(block, dict) and block.get("type") == "text":
+ existing = block.get("text", "") or ""
+ new_block = {**block, "text": f"{prefix}{existing}"}
+ return [*system[:idx], new_block, *system[idx + 1 :]]
+ return [{"type": "text", "text": prefix.rstrip()}, *system]
+
+
+def _resolve_trigger_tokens(edit_spec: Dict[str, Any]) -> Tuple[int, List[str]]:
+ """Validate and resolve ``trigger.value``.
+
+ Raises ``AnthropicContextManagementError`` if the explicitly-supplied value
+ is below the 50k minimum. Unknown ``trigger.type`` values fall back to
+ ``input_tokens`` with a warning.
+ """
+ warnings: List[str] = []
+ trigger = edit_spec.get("trigger") or {}
+ if not isinstance(trigger, dict):
+ warnings.append("trigger_not_a_dict_using_default")
+ return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
+
+ trigger_type = trigger.get("type", "input_tokens")
+ if trigger_type != "input_tokens":
+ warnings.append(f"unsupported_trigger_type_{trigger_type}_using_input_tokens")
+
+ value = trigger.get("value")
+ if value is None:
+ return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
+ if not isinstance(value, int):
+ warnings.append("trigger_value_not_int_using_default")
+ return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
+ if value < COMPACT_MIN_TRIGGER_TOKENS:
+ raise AnthropicContextManagementError(
+ status_code=400,
+ message=(
+ f"context_management.compact_20260112.trigger.value must be at "
+ f"least {COMPACT_MIN_TRIGGER_TOKENS} tokens"
+ ),
+ )
+ return value, warnings
+
+
+def _build_summary_prompt(
+ edit_spec: Dict[str, Any], tools: Optional[List[Dict[str, Any]]]
+) -> str:
+ custom = edit_spec.get("instructions")
+ if isinstance(custom, str) and custom.strip():
+ return custom
+ prompt = COMPACT_DEFAULT_INSTRUCTIONS
+ if tools:
+ prompt = f"{prompt}{COMPACT_NO_TOOL_CALLS_SUFFIX}"
+ return prompt
+
+
+def _propagate_metadata(parent_metadata: Optional[Dict[str, Any]]) -> Dict[str, Any]:
+ if not parent_metadata:
+ return {}
+ propagated: Dict[str, Any] = {}
+ for key in _PROPAGATED_METADATA_KEYS:
+ if key in parent_metadata:
+ propagated[key] = parent_metadata[key]
+ return propagated
+
+
+def _count_effective_tokens(
+ model: str,
+ effective_messages: List[Dict[str, Any]],
+ compaction_block: Optional[Dict[str, Any]],
+ tools: Optional[List[Dict[str, Any]]],
+) -> int:
+ """Token-count the conversation as it will appear downstream.
+
+ The compaction block (if any) becomes a system prefix on the downstream
+ call, so its content still counts even though it isn't in ``messages``.
+ """
+ # Local import to avoid pulling the adapter at module load time.
+ from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
+ LiteLLMAnthropicMessagesAdapter,
+ )
+
+ messages_without_compaction = _strip_compaction_blocks(effective_messages)
+ try:
+ openai_shape = (
+ LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
+ messages=cast(Any, messages_without_compaction)
+ )
+ )
+ except Exception as e:
+ verbose_logger.debug(
+ "compact_20260112: anthropic→openai translation failed during token "
+ "count, falling back to raw messages: %s",
+ e,
+ )
+ openai_shape = cast(Any, messages_without_compaction)
+
+ total = litellm.token_counter(
+ model=model,
+ messages=cast(Any, openai_shape),
+ tools=cast(Any, tools),
+ )
+ if compaction_block is not None:
+ content = compaction_block.get("content") or ""
+ if content:
+ total += litellm.token_counter(model=model, text=content)
+ return total
+
+
+def _extract_summary_text(raw: Optional[str]) -> Optional[str]:
+ if not raw:
+ return None
+ match = _SUMMARY_TAG_RE.search(raw)
+ if match is None:
+ return None
+ summary = match.group(1).strip()
+ return summary or None
+
+
+def _build_summary_messages(
+ effective_messages: List[Dict[str, Any]],
+ prompt: str,
+) -> List[Dict[str, Any]]:
+ """Build the OpenAI-shape message list for the summary call.
+
+ The conversation history is translated to OpenAI shape; the
+ summarization prompt is appended as a final user turn.
+ """
+ from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
+ LiteLLMAnthropicMessagesAdapter,
+ )
+
+ stripped = _strip_compaction_blocks(effective_messages)
+ try:
+ openai_messages = (
+ LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
+ messages=cast(Any, stripped)
+ )
+ )
+ except Exception as e:
+ verbose_logger.warning(
+ "compact_20260112: anthropic→openai translation failed when "
+ "building summary call; falling back to raw shape: %s",
+ e,
+ )
+ openai_messages = cast(Any, stripped)
+
+ return [*openai_messages, {"role": "user", "content": prompt}]
+
+
+async def _call_summary_model(
+ *,
+ summary_model: str,
+ summary_messages: List[Dict[str, Any]],
+ metadata: Dict[str, Any],
+ llm_router: Any,
+) -> Any:
+ """Invoke the configured summary model.
+
+ Prefers ``llm_router.acompletion`` so the model alias resolves against the
+ proxy's ``model_list``; falls back to ``litellm.acompletion`` if no router
+ is available (e.g. SDK usage outside the proxy).
+ """
+ call_kwargs: Dict[str, Any] = {
+ "model": summary_model,
+ "messages": summary_messages,
+ "metadata": metadata,
+ }
+ if llm_router is not None and hasattr(llm_router, "acompletion"):
+ return await llm_router.acompletion(**call_kwargs)
+ return await litellm.acompletion(**call_kwargs)
+
+
+def _extract_response_text(response: Any) -> Optional[str]:
+ try:
+ choice = response.choices[0]
+ message = choice.message
+ content = getattr(message, "content", None)
+ if isinstance(content, str):
+ return content
+ # Some providers return a list of content parts.
+ if isinstance(content, list):
+ text_parts = [
+ part.get("text", "")
+ for part in content
+ if isinstance(part, dict) and part.get("type") == "text"
+ ]
+ return "".join(text_parts) or None
+ except (AttributeError, IndexError, KeyError):
+ return None
+ return None
+
+
+def _extract_usage(response: Any) -> Tuple[int, int]:
+ usage = getattr(response, "usage", None)
+ if usage is None:
+ return 0, 0
+ return (
+ int(getattr(usage, "prompt_tokens", 0) or 0),
+ int(getattr(usage, "completion_tokens", 0) or 0),
+ )
+
+
+async def apply_compact_20260112(
+ *,
+ model: str,
+ messages: List[Dict[str, Any]],
+ tools: Optional[List[Dict[str, Any]]],
+ system: Optional[Union[str, List[Dict[str, Any]]]],
+ edit_spec: Dict[str, Any],
+ metadata: Optional[Dict[str, Any]] = None,
+ llm_router: Any = None,
+) -> PolyfillResult:
+ """Apply ``compact_20260112``; return a ``PolyfillResult``.
+
+ See module docstring for the algorithm. Errors are best-effort: when the
+ summary call fails or the response is malformed, the editor returns the
+ pre-summary state (with ``applied_edits[0].error`` populated) so the
+ original request still proceeds.
+ """
+ # Validation runs first. Raising AnthropicContextManagementError here is
+ # the only path on which the polyfill aborts the request.
+ trigger_tokens, warnings = _resolve_trigger_tokens(edit_spec)
+ if edit_spec.get("pause_after_compaction"):
+ warnings.append("pause_after_compaction_ignored")
+
+ applied: AppliedEdit = {"type": COMPACT_EDIT_TYPE}
+ if warnings:
+ applied["warnings"] = warnings
+
+ # Opt-in gate: no summary model configured → no-op.
+ summary_model = _read_summary_model_setting()
+ if summary_model is None:
+ applied["error"] = "summary_model_not_configured"
+ return PolyfillResult(
+ messages=messages,
+ system=system,
+ applied_edits=[applied],
+ )
+
+ # Phase A: slice around any existing compaction block.
+ effective_messages, prior_compaction_block = _slice_around_compaction_block(
+ messages
+ )
+ prior_summary_text = (
+ prior_compaction_block.get("content") if prior_compaction_block else None
+ )
+ augmented_system: Union[str, List[Dict[str, Any]], None] = system
+ if isinstance(prior_summary_text, str) and prior_summary_text:
+ augmented_system = _augment_system_with_summary(system, prior_summary_text)
+
+ downstream_messages = _strip_compaction_blocks(effective_messages)
+
+ # Phase B: threshold check.
+ try:
+ current_tokens = _count_effective_tokens(
+ model=model,
+ effective_messages=effective_messages,
+ compaction_block=prior_compaction_block,
+ tools=tools,
+ )
+ except Exception as e:
+ verbose_logger.warning(
+ "compact_20260112: token_counter failed; assuming under threshold: %s", e
+ )
+ current_tokens = 0
+
+ verbose_logger.debug(
+ "compact_20260112: current_tokens=%s trigger=%s", current_tokens, trigger_tokens
+ )
+
+ if current_tokens <= trigger_tokens:
+ # Slice-only path. If Phase A fired we still slice + system-prefix.
+ return PolyfillResult(
+ messages=downstream_messages,
+ system=augmented_system,
+ applied_edits=[applied],
+ )
+
+ # Phase C: summarize.
+ prompt = _build_summary_prompt(edit_spec, tools)
+ summary_messages = _build_summary_messages(effective_messages, prompt)
+ propagated_metadata = _propagate_metadata(metadata)
+
+ try:
+ response = await _call_summary_model(
+ summary_model=summary_model,
+ summary_messages=summary_messages,
+ metadata=propagated_metadata,
+ llm_router=llm_router,
+ )
+ except Exception as e:
+ verbose_logger.warning("compact_20260112: summary call failed: %s", e)
+ applied["error"] = "summary_call_failed"
+ return PolyfillResult(
+ messages=downstream_messages,
+ system=augmented_system,
+ applied_edits=[applied],
+ )
+
+ summary_text = _extract_summary_text(_extract_response_text(response))
+ if summary_text is None:
+ applied["error"] = "summary_extraction_failed"
+ return PolyfillResult(
+ messages=downstream_messages,
+ system=augmented_system,
+ applied_edits=[applied],
+ )
+
+ summary_input_tokens, summary_output_tokens = _extract_usage(response)
+ applied["summary_input_tokens"] = summary_input_tokens
+ applied["summary_output_tokens"] = summary_output_tokens
+
+ compaction_block: CompactionBlock = {
+ "type": "compaction",
+ "content": summary_text,
+ }
+ iterations_usage: List[UsageIteration] = [
+ {
+ "type": "compaction",
+ "input_tokens": summary_input_tokens,
+ "output_tokens": summary_output_tokens,
+ }
+ ]
+
+ # Per Anthropic's contract, everything before the compaction block is
+ # dropped. Phase D: the user/assistant log goes empty; the summary lives
+ # on the system message instead. Anthropic requires a non-empty messages
+ # array, so keep the most recent original user turn so the model has the
+ # question to answer.
+ summarized_system = _augment_system_with_summary(system, summary_text)
+ downstream_messages_after_summary: List[Dict[str, Any]] = []
+ for msg in reversed(messages):
+ if msg.get("role") == "user":
+ downstream_messages_after_summary = [msg]
+ break
+
+ return PolyfillResult(
+ messages=downstream_messages_after_summary,
+ system=summarized_system,
+ applied_edits=[applied],
+ compaction_block=compaction_block,
+ iterations_usage=iterations_usage,
+ )
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py b/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py
new file mode 100644
index 00000000000..1b14089a451
--- /dev/null
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py
@@ -0,0 +1,14 @@
+"""Exceptions raised by the context_management polyfill."""
+
+
+class AnthropicContextManagementError(Exception):
+ """Validation error from the polyfill, surfaced as an Anthropic-format 4xx.
+
+ The `/v1/messages` endpoint catches this in its exception handler and
+ emits an Anthropic-shaped error body instead of the default OpenAI shape.
+ """
+
+ def __init__(self, *, status_code: int, message: str) -> None:
+ super().__init__(message)
+ self.status_code = status_code
+ self.message = message
diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/result.py b/litellm/llms/anthropic/experimental_pass_through/context_management/result.py
new file mode 100644
index 00000000000..28641a90993
--- /dev/null
+++ b/litellm/llms/anthropic/experimental_pass_through/context_management/result.py
@@ -0,0 +1,24 @@
+"""``PolyfillResult`` — the shape returned by the context-management dispatcher.
+
+Threaded from the dispatcher through ``async_anthropic_messages_handler`` into
+the adapter so it can prepend the ``compaction`` block to the response and
+attach ``iterations`` to ``usage``.
+"""
+
+from dataclasses import dataclass, field
+from typing import Any, Dict, List, Optional, Union
+
+from litellm.types.llms.anthropic import (
+ AppliedEdit,
+ CompactionBlock,
+ UsageIteration,
+)
+
+
+@dataclass
+class PolyfillResult:
+ messages: List[Dict[str, Any]]
+ system: Optional[Union[str, List[Dict[str, Any]]]]
+ applied_edits: List[AppliedEdit] = field(default_factory=list)
+ compaction_block: Optional[CompactionBlock] = None
+ iterations_usage: Optional[List[UsageIteration]] = None
diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py
index 2551353a049..2210caa5a79 100644
--- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py
+++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py
@@ -459,40 +459,12 @@ def anthropic_messages_handler(
**_shared_kwargs
)
- # In-gateway context_management polyfill on the chat-completions adapter
- # (native on Anthropic/Responses paths). Skipped when drop_params is on.
- context_management_spec = _shared_kwargs.pop("context_management", None)
- _request_drop_params = _shared_kwargs.get("drop_params")
- _drop_params = (
- _request_drop_params
- if _request_drop_params is not None
- else litellm.drop_params
- )
- polyfill_applied_edits: Optional[List[AppliedEdit]] = None
- if context_management_spec and not _drop_params:
- from litellm.llms.anthropic.experimental_pass_through.context_management import (
- apply_context_management,
- )
-
- try:
- edited_messages, polyfill_applied_edits = apply_context_management(
- model=model,
- messages=_shared_kwargs["messages"],
- tools=_shared_kwargs.get("tools"),
- system=_shared_kwargs.get("system"),
- context_management_spec=context_management_spec,
- )
- _shared_kwargs["messages"] = edited_messages
- except Exception as e:
- verbose_logger.exception(
- "context_management polyfill: skipping edits due to error: %s",
- e,
- )
- polyfill_applied_edits = None
-
+ # The in-gateway context_management polyfill runs inside
+ # ``async_anthropic_messages_handler`` so it can ``await`` the
+ # summarization model for ``compact_20260112``. ``context_management``
+ # is passed through as a regular kwarg.
return (
LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(
- _polyfill_applied_edits=polyfill_applied_edits,
**_shared_kwargs,
)
)
diff --git a/litellm/proxy/anthropic_endpoints/endpoints.py b/litellm/proxy/anthropic_endpoints/endpoints.py
index 69d69354fd1..79e2648744b 100644
--- a/litellm/proxy/anthropic_endpoints/endpoints.py
+++ b/litellm/proxy/anthropic_endpoints/endpoints.py
@@ -3,10 +3,14 @@ Unified /v1/messages endpoint - (Anthropic Spec)
"""
from fastapi import APIRouter, Depends, HTTPException, Request, Response
+from fastapi.responses import JSONResponse
from litellm._logging import verbose_proxy_logger
from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping
from litellm.integrations.custom_guardrail import ModifyResponseException
+from litellm.llms.anthropic.experimental_pass_through.context_management import (
+ AnthropicContextManagementError,
+)
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_request_processing import (
@@ -114,6 +118,13 @@ async def anthropic_response( # noqa: PLR0915
)
return _anthropic_response
+ except AnthropicContextManagementError as e:
+ body = AnthropicExceptionMapping.transform_to_anthropic_error(
+ status_code=e.status_code,
+ raw_message=e.message,
+ request_id=request.headers.get("x-request-id"),
+ )
+ return JSONResponse(status_code=e.status_code, content=body)
except Exception as e:
await proxy_logging_obj.post_call_failure_hook(
user_api_key_dict=user_api_key_dict, original_exception=e, request_data=data
diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py
index 4655951ae24..2fb881cb003 100644
--- a/litellm/types/llms/anthropic.py
+++ b/litellm/types/llms/anthropic.py
@@ -521,6 +521,11 @@ class AppliedEdit(TypedDict, total=False):
cleared_input_tokens: int
cleared_tool_uses: int
cleared_thinking_turns: int
+ # compact_20260112 fields
+ summary_input_tokens: int
+ summary_output_tokens: int
+ error: str
+ warnings: List[str]
class ContextManagementResponse(TypedDict, total=False):
@@ -529,6 +534,21 @@ class ContextManagementResponse(TypedDict, total=False):
applied_edits: List[AppliedEdit]
+class CompactionBlock(TypedDict, total=False):
+ """Synthesized ``compaction`` content block (compact_20260112)."""
+
+ type: Literal["compaction"]
+ content: Optional[str]
+
+
+class UsageIteration(TypedDict, total=False):
+ """One sampling iteration's token usage (compact_20260112)."""
+
+ type: Literal["compaction", "message"]
+ input_tokens: int
+ output_tokens: int
+
+
class MessageBlockDelta(TypedDict):
"""
Anthropic
diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
index 44530fecebd..74e1e17e6d7 100644
--- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
+++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
@@ -2472,3 +2472,172 @@ def test_translate_anthropic_tool_choice_none():
result = adapter.translate_anthropic_tool_choice_to_openai({"type": "none"})
assert result == "none"
+
+
+# ---------------------------------------------------------------------------
+# PolyfillResult integration tests
+# ---------------------------------------------------------------------------
+
+
+def _make_simple_openai_response(
+ text: str = "Hello", prompt_tokens: int = 10, completion_tokens: int = 5
+) -> ModelResponse:
+ return ModelResponse(
+ id="resp_polyfill_test",
+ model="gpt-4o",
+ choices=[
+ Choices(
+ finish_reason="stop",
+ message=Message(role="assistant", content=text),
+ )
+ ],
+ usage=Usage(prompt_tokens=prompt_tokens, completion_tokens=completion_tokens),
+ )
+
+
+def test_translate_openai_response_to_anthropic_with_polyfill_compaction_block():
+ """compaction_block from PolyfillResult must be prepended to content at index 0."""
+ from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
+ PolyfillResult,
+ )
+
+ compaction_block = {"type": "compaction", "content": "Summary of prior turns."}
+ polyfill = PolyfillResult(
+ messages=[],
+ system=None,
+ applied_edits=[{"type": "compact_20260112"}],
+ compaction_block=compaction_block,
+ iterations_usage=None,
+ )
+ response = _make_simple_openai_response(text="Hello after compaction.")
+ adapter = LiteLLMAnthropicMessagesAdapter()
+ result = adapter.translate_openai_response_to_anthropic(
+ response=response, polyfill_result=polyfill
+ )
+
+ content = result.get("content")
+ assert content is not None
+ assert content[0]["type"] == "compaction"
+ assert content[0]["content"] == "Summary of prior turns."
+ assert content[1]["type"] == "text"
+ assert content[1]["text"] == "Hello after compaction."
+
+ # applied_edits must surface on context_management
+ cm = result.get("context_management")
+ assert cm is not None
+ assert cm["applied_edits"][0]["type"] == "compact_20260112"
+
+
+def test_translate_openai_response_to_anthropic_with_polyfill_iterations_usage():
+ """iterations_usage from PolyfillResult must produce usage['iterations'] with a message entry."""
+ from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
+ PolyfillResult,
+ )
+
+ polyfill = PolyfillResult(
+ messages=[],
+ system=None,
+ applied_edits=[{"type": "compact_20260112"}],
+ compaction_block=None,
+ iterations_usage=[
+ {"type": "compaction", "input_tokens": 200, "output_tokens": 50},
+ ],
+ )
+ response = _make_simple_openai_response(prompt_tokens=100, completion_tokens=30)
+ adapter = LiteLLMAnthropicMessagesAdapter()
+ result = adapter.translate_openai_response_to_anthropic(
+ response=response, polyfill_result=polyfill
+ )
+
+ usage = result.get("usage")
+ assert usage is not None
+ iterations = usage.get("iterations")
+ assert iterations is not None
+ assert len(iterations) == 2
+ assert iterations[0] == {
+ "type": "compaction",
+ "input_tokens": 200,
+ "output_tokens": 50,
+ }
+ assert iterations[1]["type"] == "message"
+ assert iterations[1]["input_tokens"] == 100
+ assert iterations[1]["output_tokens"] == 30
+
+ # Top-level tokens must still reflect the message iteration
+ assert usage["input_tokens"] == 100
+ assert usage["output_tokens"] == 30
+
+
+def test_translate_openai_response_to_anthropic_no_polyfill_no_change():
+ """Without a PolyfillResult the response must be unchanged (no compaction, no iterations)."""
+ response = _make_simple_openai_response()
+ adapter = LiteLLMAnthropicMessagesAdapter()
+ result = adapter.translate_openai_response_to_anthropic(response=response)
+
+ content = result.get("content")
+ assert content is not None
+ assert content[0]["type"] == "text"
+
+ usage = result.get("usage")
+ assert usage is not None
+ assert "iterations" not in usage
+
+
+def test_translate_openai_response_to_anthropic_with_polyfill_both_compaction_and_iterations():
+ """Full summary path: compaction_block and iterations_usage both present simultaneously."""
+ from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
+ PolyfillResult,
+ )
+
+ compaction_block = {
+ "type": "compaction",
+ "content": "Summary of a long conversation.",
+ }
+ polyfill = PolyfillResult(
+ messages=[],
+ system=None,
+ applied_edits=[{"type": "compact_20260112"}],
+ compaction_block=compaction_block,
+ iterations_usage=[
+ {"type": "compaction", "input_tokens": 300, "output_tokens": 75},
+ ],
+ )
+ response = _make_simple_openai_response(
+ text="After compaction.", prompt_tokens=120, completion_tokens=40
+ )
+ adapter = LiteLLMAnthropicMessagesAdapter()
+ result = adapter.translate_openai_response_to_anthropic(
+ response=response, polyfill_result=polyfill
+ )
+
+ # compaction block must come first
+ content = result.get("content")
+ assert content is not None
+ assert content[0]["type"] == "compaction"
+ assert content[0]["content"] == "Summary of a long conversation."
+ assert content[1]["type"] == "text"
+ assert content[1]["text"] == "After compaction."
+
+ # iterations: compaction entry + message entry
+ usage = result.get("usage")
+ assert usage is not None
+ iterations = usage.get("iterations")
+ assert iterations is not None
+ assert len(iterations) == 2
+ assert iterations[0] == {
+ "type": "compaction",
+ "input_tokens": 300,
+ "output_tokens": 75,
+ }
+ assert iterations[1]["type"] == "message"
+ assert iterations[1]["input_tokens"] == 120
+ assert iterations[1]["output_tokens"] == 40
+
+ # top-level tokens match the message iteration
+ assert usage["input_tokens"] == 120
+ assert usage["output_tokens"] == 40
+
+ # context_management applied_edits must surface
+ cm = result.get("context_management")
+ assert cm is not None
+ assert cm["applied_edits"][0]["type"] == "compact_20260112"
diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py
new file mode 100644
index 00000000000..96e5f23fc8c
--- /dev/null
+++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py
@@ -0,0 +1,862 @@
+"""
+Unit tests for the compact_20260112 polyfill editor.
+
+Coverage:
+- trigger.value < 50k → AnthropicContextManagementError(400)
+- opt-in gate (no summary model) → summary_model_not_configured
+- slice-only path (existing compaction block, under threshold)
+- full summary path (over threshold, summary fires)
+- summary call raises → summary_call_failed
+- summary response missing tags → summary_extraction_failed
+- pause_after_compaction: true → pause_after_compaction_ignored warning, proceeds
+- custom instructions → default prompt is not used even when tools present
+"""
+
+from typing import Any, Dict, List
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
+
+from litellm.llms.anthropic.experimental_pass_through.context_management import (
+ AnthropicContextManagementError,
+ apply_context_management,
+)
+from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import (
+ _augment_system_with_summary,
+ _extract_summary_text,
+ _slice_around_compaction_block,
+ _strip_compaction_blocks,
+ apply_compact_20260112,
+)
+
+MODEL = "openai/gpt-4o"
+
+_EDIT_SPEC_DEFAULT: Dict[str, Any] = {"type": "compact_20260112"}
+
+
+# ---------------------------------------------------------------------------
+# Helpers
+# ---------------------------------------------------------------------------
+
+
+def _simple_messages() -> List[Dict[str, Any]]:
+ return [
+ {"role": "user", "content": "Hello"},
+ {"role": "assistant", "content": [{"type": "text", "text": "Hi there"}]},
+ {"role": "user", "content": "What is 2+2?"},
+ ]
+
+
+def _messages_with_compaction(summary: str = "prev summary") -> List[Dict[str, Any]]:
+ """History that already has a compaction block in an assistant turn."""
+ return [
+ {"role": "user", "content": "older question"},
+ {
+ "role": "assistant",
+ "content": [{"type": "compaction", "content": summary}],
+ },
+ {"role": "user", "content": "newer question"},
+ {"role": "assistant", "content": [{"type": "text", "text": "newer reply"}]},
+ {"role": "user", "content": "latest question"},
+ ]
+
+
+def _make_mock_response(
+ content: str,
+ prompt_tokens: int = 50,
+ completion_tokens: int = 100,
+) -> MagicMock:
+ response = MagicMock()
+ choice = MagicMock()
+ message = MagicMock()
+ message.content = content
+ choice.message = message
+ response.choices = [choice]
+ usage = MagicMock()
+ usage.prompt_tokens = prompt_tokens
+ usage.completion_tokens = completion_tokens
+ response.usage = usage
+ return response
+
+
+# ---------------------------------------------------------------------------
+# Unit: helper functions
+# ---------------------------------------------------------------------------
+
+
+def test_slice_around_compaction_block_found():
+ messages = _messages_with_compaction("my summary")
+ sliced, block = _slice_around_compaction_block(messages)
+ assert block is not None
+ assert block["type"] == "compaction"
+ assert block["content"] == "my summary"
+ # Sliced list starts at the assistant turn containing the compaction block
+ assert sliced[0]["role"] == "assistant"
+ assert len(sliced) == 4 # assistant(compaction), user, assistant, user
+
+
+def test_slice_around_compaction_block_not_found():
+ messages = _simple_messages()
+ sliced, block = _slice_around_compaction_block(messages)
+ assert block is None
+ assert sliced is messages # same object, no copy
+
+
+def test_strip_compaction_blocks_removes_block():
+ messages = [
+ {
+ "role": "assistant",
+ "content": [
+ {"type": "compaction", "content": "summary"},
+ {"type": "text", "text": "hello"},
+ ],
+ }
+ ]
+ stripped = _strip_compaction_blocks(messages)
+ assert len(stripped) == 1
+ content = stripped[0]["content"]
+ assert all(b["type"] != "compaction" for b in content)
+ assert len(content) == 1
+ assert content[0]["type"] == "text"
+
+
+def test_strip_compaction_blocks_drops_compaction_only_turn():
+ messages = [
+ {"role": "user", "content": "hi"},
+ {
+ "role": "assistant",
+ "content": [{"type": "compaction", "content": "summary"}],
+ },
+ {"role": "user", "content": "bye"},
+ ]
+ stripped = _strip_compaction_blocks(messages)
+ assert len(stripped) == 2
+ assert stripped[0]["role"] == "user"
+ assert stripped[1]["role"] == "user"
+
+
+def test_augment_system_with_summary_none_system():
+ result = _augment_system_with_summary(None, "my summary")
+ assert isinstance(result, str)
+ assert "my summary" in result
+
+
+def test_augment_system_with_summary_string_system():
+ result = _augment_system_with_summary("You are helpful.", "my summary")
+ assert isinstance(result, str)
+ assert result.startswith("Previous conversation summary:")
+ assert "my summary" in result
+ assert "You are helpful." in result
+
+
+def test_augment_system_with_summary_list_system():
+ system = [{"type": "text", "text": "existing system"}]
+ result = _augment_system_with_summary(system, "my summary")
+ assert isinstance(result, list)
+ assert result[0]["type"] == "text"
+ text = result[0]["text"]
+ assert "my summary" in text
+ assert "existing system" in text
+
+
+def test_extract_summary_text_found():
+ raw = "Here is the summary:\nKey points from chat\nDone."
+ assert _extract_summary_text(raw) == "Key points from chat"
+
+
+def test_extract_summary_text_missing_tags():
+ assert _extract_summary_text("No tags here") is None
+
+
+def test_extract_summary_text_none():
+ assert _extract_summary_text(None) is None
+
+
+def test_extract_summary_text_case_insensitive():
+ raw = "uppercase tags"
+ assert _extract_summary_text(raw) == "uppercase tags"
+
+
+# ---------------------------------------------------------------------------
+# Editor: validation
+# ---------------------------------------------------------------------------
+
+
+async def test_trigger_below_minimum_raises():
+ with pytest.raises(AnthropicContextManagementError) as exc_info:
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=_simple_messages(),
+ tools=None,
+ system=None,
+ edit_spec={
+ "type": "compact_20260112",
+ "trigger": {"type": "input_tokens", "value": 10_000},
+ },
+ )
+ assert exc_info.value.status_code == 400
+ assert "50000" in exc_info.value.message
+
+
+async def test_trigger_at_minimum_does_not_raise():
+ """Exactly 50 000 is allowed — only strictly less than 50k is rejected."""
+ with patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value=None,
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=_simple_messages(),
+ tools=None,
+ system=None,
+ edit_spec={
+ "type": "compact_20260112",
+ "trigger": {"type": "input_tokens", "value": 50_000},
+ },
+ )
+ # Reached opt-in gate (no summary model); no error raised from trigger check
+ assert result.applied_edits[0]["error"] == "summary_model_not_configured"
+
+
+# ---------------------------------------------------------------------------
+# Editor: opt-in gate
+# ---------------------------------------------------------------------------
+
+
+async def test_opt_in_gating_no_summary_model_configured():
+ messages = _simple_messages()
+ with patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value=None,
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system="system prompt",
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+ assert result.applied_edits[0]["error"] == "summary_model_not_configured"
+ assert result.messages == messages
+ assert result.system == "system prompt"
+ assert result.compaction_block is None
+ assert result.iterations_usage is None
+
+
+# ---------------------------------------------------------------------------
+# Editor: slice-only path
+# ---------------------------------------------------------------------------
+
+
+async def test_slice_only_path_with_existing_compaction_block():
+ """Phase A slices; Phase B token count is below threshold; no summary call."""
+ messages = _messages_with_compaction("prior summary text")
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=500), # well under threshold
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ # System should have the prior summary prefixed
+ assert result.system is not None
+ assert "prior summary text" in str(result.system)
+
+ # No new compaction block; no iterations_usage
+ assert result.compaction_block is None
+ assert result.iterations_usage is None
+
+ # No compaction block in downstream messages
+ for msg in result.messages:
+ content = msg.get("content")
+ if isinstance(content, list):
+ for block in content:
+ assert block.get("type") != "compaction"
+
+
+async def test_slice_only_no_compaction_block_under_threshold():
+ """No prior compaction block, and token count is below threshold — pure pass-through."""
+ messages = _simple_messages()
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=500),
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ assert result.messages == messages
+ assert result.compaction_block is None
+ assert result.iterations_usage is None
+ assert not result.applied_edits[0].get("error")
+
+
+# ---------------------------------------------------------------------------
+# Editor: full summary path
+# ---------------------------------------------------------------------------
+
+
+async def test_full_summary_path():
+ """Over threshold: summary call fires, compaction_block and iterations_usage returned."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response(
+ "Condensed history", prompt_tokens=200, completion_tokens=50
+ )
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000), # over 150k threshold
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ new_callable=AsyncMock,
+ return_value=mock_response,
+ ),
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ assert result.compaction_block is not None
+ assert result.compaction_block["type"] == "compaction"
+ assert result.compaction_block["content"] == "Condensed history"
+
+ assert result.iterations_usage is not None
+ assert len(result.iterations_usage) == 1
+ assert result.iterations_usage[0]["type"] == "compaction"
+ assert result.iterations_usage[0]["input_tokens"] == 200
+ assert result.iterations_usage[0]["output_tokens"] == 50
+
+ # System must have summary prefixed
+ assert "Condensed history" in str(result.system)
+
+ # applied_edits should have usage fields
+ edit = result.applied_edits[0]
+ assert edit["type"] == "compact_20260112"
+ assert edit.get("summary_input_tokens") == 200
+ assert edit.get("summary_output_tokens") == 50
+
+ # Downstream messages must not contain a compaction block
+ for msg in result.messages:
+ content = msg.get("content")
+ if isinstance(content, list):
+ for block in content:
+ assert block.get("type") != "compaction"
+
+
+async def test_full_summary_path_uses_router_when_available():
+ """When llm_router is provided, its acompletion method is called instead of litellm."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("Router summary")
+ mock_router = MagicMock()
+ mock_router.acompletion = AsyncMock(return_value=mock_response)
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="my-summary-model",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ llm_router=mock_router,
+ )
+
+ mock_router.acompletion.assert_called_once()
+ call_kwargs = mock_router.acompletion.call_args.kwargs
+ assert call_kwargs["model"] == "my-summary-model"
+
+ assert result.compaction_block is not None
+ assert result.compaction_block["content"] == "Router summary"
+
+
+async def test_metadata_propagated_to_summary_call():
+ """Auth metadata from the parent request is forwarded to the summary call."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("Summary")
+ parent_metadata = {
+ "user_api_key": "sk-test",
+ "user_api_key_team_id": "team-123",
+ "user_api_key_user_id": "user-456",
+ "litellm_call_id": "call-789",
+ "should_not_propagate": "secret",
+ }
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ new_callable=AsyncMock,
+ return_value=mock_response,
+ ) as mock_call,
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ metadata=parent_metadata,
+ )
+
+ call_kwargs = mock_call.call_args.kwargs
+ propagated = call_kwargs["metadata"]
+ assert propagated["user_api_key"] == "sk-test"
+ assert propagated["user_api_key_team_id"] == "team-123"
+ assert "should_not_propagate" not in propagated
+
+
+# ---------------------------------------------------------------------------
+# Editor: error paths
+# ---------------------------------------------------------------------------
+
+
+async def test_summary_call_failed():
+ """When the summary model raises, applied_edits[0].error == 'summary_call_failed'."""
+ messages = _simple_messages()
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ new_callable=AsyncMock,
+ side_effect=RuntimeError("network error"),
+ ),
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ assert result.applied_edits[0]["error"] == "summary_call_failed"
+ assert result.compaction_block is None
+ assert result.iterations_usage is None
+ # Messages passed through (at minimum sliced, no compaction blocks)
+ for msg in result.messages:
+ content = msg.get("content")
+ if isinstance(content, list):
+ for block in content:
+ assert block.get("type") != "compaction"
+
+
+async def test_summary_extraction_failed_no_tags():
+ """When summary response has no tags, applied_edits[0].error == 'summary_extraction_failed'."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("I cannot summarize that.")
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ new_callable=AsyncMock,
+ return_value=mock_response,
+ ),
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ assert result.applied_edits[0]["error"] == "summary_extraction_failed"
+ assert result.compaction_block is None
+ assert result.iterations_usage is None
+
+
+# ---------------------------------------------------------------------------
+# Editor: warnings
+# ---------------------------------------------------------------------------
+
+
+async def test_pause_after_compaction_ignored_warning():
+ """pause_after_compaction: true → warning recorded, request proceeds normally."""
+ messages = _simple_messages()
+ with patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value=None,
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec={
+ "type": "compact_20260112",
+ "pause_after_compaction": True,
+ },
+ )
+
+ edit = result.applied_edits[0]
+ assert "pause_after_compaction_ignored" in (edit.get("warnings") or [])
+ # Request still proceeds (here it hits opt-in gate because no model configured)
+ assert edit.get("error") == "summary_model_not_configured"
+
+
+async def test_unsupported_trigger_type_falls_back_to_default():
+ messages = _simple_messages()
+ with patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value=None,
+ ):
+ result = await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec={
+ "type": "compact_20260112",
+ "trigger": {"type": "output_tokens", "value": 200_000},
+ },
+ )
+
+ edit = result.applied_edits[0]
+ warnings = edit.get("warnings") or []
+ assert any("unsupported_trigger_type" in w for w in warnings)
+
+
+# ---------------------------------------------------------------------------
+# Editor: custom instructions
+# ---------------------------------------------------------------------------
+
+
+async def test_custom_instructions_used_verbatim():
+ """Custom instructions are used as-is; the default prompt is NOT appended."""
+ messages = _simple_messages()
+ tools = [{"name": "search", "description": "Search tool"}]
+ mock_response = _make_mock_response("Custom summary")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=tools,
+ system=None,
+ edit_spec={
+ "type": "compact_20260112",
+ "instructions": "Summarize everything briefly.",
+ },
+ )
+
+ assert len(captured_calls) == 1
+ summary_messages = captured_calls[0]["summary_messages"]
+ # The last message should be the custom instruction prompt
+ last_msg = summary_messages[-1]
+ assert last_msg["role"] == "user"
+ assert last_msg["content"] == "Summarize everything briefly."
+ # The "do not call tools" suffix should NOT be in the prompt since custom was set
+ assert "tool" not in last_msg["content"].lower()
+
+
+async def test_default_instructions_appended_with_no_tool_suffix_when_no_tools():
+ """Without tools, default prompt is used but the no-tool-calls suffix is absent."""
+ messages = _simple_messages()
+ mock_response = _make_mock_response("Default summary")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ prompt = captured_calls[0]["summary_messages"][-1]["content"]
+ # Should not contain the no-tool-calls guidance
+ assert "do not call" not in prompt.lower()
+
+
+async def test_default_instructions_with_tools_appends_no_tool_suffix():
+ """With tools and no custom instructions, the no-tool-calls suffix is appended."""
+ messages = _simple_messages()
+ tools = [{"name": "search"}]
+ mock_response = _make_mock_response("Tool-aware summary")
+
+ captured_calls: list = []
+
+ async def _fake_call_summary_model(**kwargs):
+ captured_calls.append(kwargs)
+ return mock_response
+
+ with (
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value="claude-haiku-4-5",
+ ),
+ patch("litellm.token_counter", return_value=200_000),
+ patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
+ side_effect=_fake_call_summary_model,
+ ),
+ ):
+ await apply_compact_20260112(
+ model=MODEL,
+ messages=messages,
+ tools=tools,
+ system=None,
+ edit_spec=_EDIT_SPEC_DEFAULT,
+ )
+
+ prompt = captured_calls[0]["summary_messages"][-1]["content"]
+ assert "tool" in prompt.lower()
+
+
+# ---------------------------------------------------------------------------
+# Dispatcher integration: compact_20260112 via apply_context_management
+# ---------------------------------------------------------------------------
+
+
+async def test_dispatcher_routes_compact_edit():
+ """compact_20260112 in the dispatcher resolves to opt-in gate when no model set."""
+ messages = _simple_messages()
+ with patch(
+ "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
+ return_value=None,
+ ):
+ result = await apply_context_management(
+ model=MODEL,
+ messages=messages,
+ tools=None,
+ system=None,
+ context_management_spec={"edits": [{"type": "compact_20260112"}]},
+ )
+
+ assert len(result.applied_edits) == 1
+ assert result.applied_edits[0]["type"] == "compact_20260112"
+ assert result.applied_edits[0].get("error") == "summary_model_not_configured"
+
+
+async def test_dispatcher_trigger_below_minimum_raises_through():
+ """AnthropicContextManagementError from the editor bubbles up through the dispatcher."""
+ with pytest.raises(AnthropicContextManagementError):
+ await apply_context_management(
+ model=MODEL,
+ messages=_simple_messages(),
+ tools=None,
+ system=None,
+ context_management_spec={
+ "edits": [
+ {
+ "type": "compact_20260112",
+ "trigger": {"type": "input_tokens", "value": 1_000},
+ }
+ ]
+ },
+ )
+
+
+# ---------------------------------------------------------------------------
+# _run_polyfill_if_enabled: drop_params gate
+# ---------------------------------------------------------------------------
+
+
+async def test_run_polyfill_skipped_when_drop_params_true():
+ """When drop_params=True the polyfill must be skipped (returns None)."""
+ from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
+ _run_polyfill_if_enabled,
+ )
+
+ result = await _run_polyfill_if_enabled(
+ model=MODEL,
+ messages=_simple_messages(),
+ tools=None,
+ system=None,
+ context_management_spec={"edits": [{"type": "compact_20260112"}]},
+ metadata={},
+ drop_params=True,
+ llm_router=None,
+ )
+ assert result is None
+
+
+async def test_run_polyfill_skipped_when_spec_empty():
+ """Empty context_management_spec must also return None (no polyfill work)."""
+ from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
+ _run_polyfill_if_enabled,
+ )
+
+ result = await _run_polyfill_if_enabled(
+ model=MODEL,
+ messages=_simple_messages(),
+ tools=None,
+ system=None,
+ context_management_spec=None,
+ metadata={},
+ drop_params=False,
+ llm_router=None,
+ )
+ assert result is None
+
+
+# ---------------------------------------------------------------------------
+# Endpoint error format: AnthropicContextManagementError → Anthropic 400 body
+# ---------------------------------------------------------------------------
+
+
+def test_anthropic_context_management_error_format():
+ """AnthropicContextManagementError must produce an Anthropic-format body via
+ AnthropicExceptionMapping.transform_to_anthropic_error — the same path the
+ /v1/messages endpoint takes when it catches this exception."""
+ from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping
+
+ body = AnthropicExceptionMapping.transform_to_anthropic_error(
+ status_code=400,
+ raw_message="trigger.value must be at least 50000 tokens",
+ request_id=None,
+ )
+
+ assert body["type"] == "error"
+ assert body["error"]["type"] == "invalid_request_error"
+ assert "50000" in body["error"]["message"]
+
+
+def test_anthropic_context_management_error_attrs():
+ """AnthropicContextManagementError carries status_code and message correctly."""
+ err = AnthropicContextManagementError(
+ status_code=400,
+ message="trigger.value must be at least 50000 tokens",
+ )
+
+ assert err.status_code == 400
+ assert "50000" in err.message
+
+
+# ---------------------------------------------------------------------------
+# Endpoint integration: /v1/messages → Anthropic 400 on context management error
+# ---------------------------------------------------------------------------
+
+
+def test_endpoint_returns_anthropic_400_on_context_management_error():
+ """The /v1/messages endpoint must catch AnthropicContextManagementError and
+ return an Anthropic-format 400 JSONResponse — not a 500 ProxyException."""
+ import sys
+ from unittest.mock import AsyncMock, MagicMock, patch
+
+ from fastapi import FastAPI
+ from fastapi.testclient import TestClient
+
+ from litellm.proxy.anthropic_endpoints.endpoints import router
+ from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
+
+ # Stub proxy_server to avoid apscheduler/heavy proxy deps imported lazily
+ # inside the route handler at request time.
+ mock_proxy_server = MagicMock()
+ mock_proxy_server.general_settings = {}
+ mock_proxy_server.llm_router = None
+ mock_proxy_server.proxy_config = MagicMock()
+ mock_proxy_server.proxy_logging_obj = MagicMock()
+ mock_proxy_server.user_api_base = None
+ mock_proxy_server.user_max_tokens = None
+ mock_proxy_server.user_model = None
+ mock_proxy_server.user_request_timeout = None
+ mock_proxy_server.user_temperature = None
+ mock_proxy_server.version = "test"
+
+ with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}):
+ with patch(
+ "litellm.proxy.anthropic_endpoints.endpoints.ProxyBaseLLMRequestProcessing"
+ ) as mock_cls:
+ mock_instance = MagicMock()
+ mock_instance.base_process_llm_request = AsyncMock(
+ side_effect=AnthropicContextManagementError(
+ status_code=400,
+ message="trigger.value must be at least 50000 tokens",
+ )
+ )
+ mock_cls.return_value = mock_instance
+
+ app = FastAPI()
+ app.include_router(router)
+ app.dependency_overrides[user_api_key_auth] = lambda: MagicMock()
+
+ client = TestClient(app, raise_server_exceptions=False)
+ response = client.post(
+ "/v1/messages",
+ json={
+ "model": "gpt-4o",
+ "messages": [{"role": "user", "content": "hi"}],
+ },
+ headers={"Authorization": "Bearer test-key"},
+ )
+
+ assert response.status_code == 400
+ body = response.json()
+ assert body["type"] == "error"
+ assert body["error"]["type"] == "invalid_request_error"
+ assert "50000" in body["error"]["message"]
diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py
index 924f79d0902..50c72cfe8d0 100644
--- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py
+++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py
@@ -43,9 +43,9 @@ def _history_with_two_tool_pairs():
]
-def test_unknown_edit_type_is_noop():
+async def test_unknown_edit_type_is_noop():
messages = _history_with_two_tool_pairs()
- new_messages, applied = apply_context_management(
+ result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@@ -54,13 +54,13 @@ def test_unknown_edit_type_is_noop():
"edits": [{"type": "totally_not_a_real_edit_20999999"}]
},
)
- assert applied == []
- assert new_messages == messages
+ assert result.applied_edits == []
+ assert result.messages == messages
-def test_known_edit_is_applied():
+async def test_known_edit_is_applied():
messages = _history_with_two_tool_pairs()
- _, applied = apply_context_management(
+ result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@@ -75,14 +75,14 @@ def test_known_edit_is_applied():
]
},
)
- assert len(applied) == 1
- assert applied[0]["type"] == "clear_tool_uses_20250919"
- assert applied[0]["cleared_tool_uses"] == 1
+ assert len(result.applied_edits) == 1
+ assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919"
+ assert result.applied_edits[0]["cleared_tool_uses"] == 1
-def test_mixed_known_unknown_only_known_applied():
+async def test_mixed_known_unknown_only_known_applied():
messages = _history_with_two_tool_pairs()
- _, applied = apply_context_management(
+ result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@@ -99,33 +99,33 @@ def test_mixed_known_unknown_only_known_applied():
]
},
)
- assert len(applied) == 1
- assert applied[0]["type"] == "clear_tool_uses_20250919"
+ assert len(result.applied_edits) == 1
+ assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919"
-def test_empty_or_missing_edits_list():
+async def test_empty_or_missing_edits_list():
messages = _history_with_two_tool_pairs()
for spec in [{}, {"edits": None}, {"edits": []}, None]:
- new_messages, applied = apply_context_management(
+ result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
system=None,
context_management_spec=spec, # type: ignore[arg-type]
)
- assert applied == []
- assert new_messages == messages
+ assert result.applied_edits == []
+ assert result.messages == messages
-def test_malformed_edit_entries_are_skipped():
+async def test_malformed_edit_entries_are_skipped():
"""Non-dict entries in `edits` list should be silently skipped."""
messages = _history_with_two_tool_pairs()
- new_messages, applied = apply_context_management(
+ result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
system=None,
context_management_spec={"edits": ["not a dict", 42, None, {"type": None}]},
)
- assert applied == []
- assert new_messages == messages
+ assert result.applied_edits == []
+ assert result.messages == messages