feat(context_management): add compact_20260112 polyfill for non-Anthropic providers

Implements an in-gateway compaction polyfill that summarizes long conversations
using a configurable model when `compact_20260112` is requested for non-Anthropic
targets (e.g. OpenAI, Gemini), matching Anthropic's context management beta
behaviour for those providers.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Sameer Kankute 2026-05-26 17:30:03 +05:30
parent 9388ea9067
commit 43637cd532
No known key found for this signature in database
15 changed files with 1860 additions and 113 deletions

View file

@ -12,13 +12,18 @@ from typing import (
)
import litellm
from litellm._logging import verbose_logger
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
AnthropicAdapter,
)
from litellm.llms.anthropic.experimental_pass_through.context_management import (
AnthropicContextManagementError,
PolyfillResult,
apply_context_management,
)
from litellm.llms.anthropic.experimental_pass_through.utils import (
is_reasoning_auto_summary_enabled,
)
from litellm.types.llms.anthropic import AppliedEdit
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
@ -31,9 +36,58 @@ if TYPE_CHECKING:
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset(
{"output_config", "context_management", "_polyfill_applied_edits"}
{"output_config", "context_management", "_polyfill_result"}
)
async def _run_polyfill_if_enabled(
*,
model: str,
messages: List[Dict],
tools: Optional[List[Dict]],
system: Optional[Any],
context_management_spec: Any,
metadata: Optional[Dict],
drop_params: Optional[bool],
llm_router: Any,
) -> Optional[PolyfillResult]:
"""Run the async context_management polyfill if a spec is present.
Returns ``None`` when the spec is empty or drop_params is on. Raises
``AnthropicContextManagementError`` so the /v1/messages endpoint can
emit an Anthropic-format 400. All other exceptions are best-effort
swallowed (matches v0 behavior).
"""
if not context_management_spec:
return None
effective_drop_params = (
drop_params if drop_params is not None else litellm.drop_params
)
if effective_drop_params:
return None
try:
return await apply_context_management(
model=model,
messages=messages,
tools=tools,
system=system,
context_management_spec=context_management_spec,
metadata=metadata,
llm_router=llm_router,
)
except AnthropicContextManagementError:
# Surface validation errors so the endpoint can emit an Anthropic-format
# 400. Other exception types fall into the best-effort branch below.
raise
except Exception as e:
verbose_logger.exception(
"context_management polyfill: skipping edits due to error: %s", e
)
return None
########################################################
# init adapter
ANTHROPIC_ADAPTER = AnthropicAdapter()
@ -303,21 +357,49 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_k: Optional[int] = None,
top_p: Optional[float] = None,
output_format: Optional[Dict] = None,
_polyfill_applied_edits: Optional[List[AppliedEdit]] = None,
**kwargs,
) -> Union[AnthropicMessagesResponse, AsyncIterator]:
"""Handle non-Anthropic models asynchronously using the adapter"""
context_management = kwargs.pop("context_management", None)
drop_params: Optional[bool] = kwargs.get("drop_params", None)
litellm_router = kwargs.pop("litellm_router", None)
if litellm_router is None:
try:
from litellm.proxy.proxy_server import llm_router as _proxy_router
litellm_router = _proxy_router
except Exception:
pass
polyfill_result = await _run_polyfill_if_enabled(
model=model,
messages=messages,
tools=tools,
system=system,
context_management_spec=context_management,
metadata=metadata,
drop_params=drop_params,
llm_router=litellm_router,
)
effective_messages = (
polyfill_result.messages if polyfill_result is not None else messages
)
effective_system = (
polyfill_result.system if polyfill_result is not None else system
)
(
completion_kwargs,
tool_name_mapping,
) = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
max_tokens=max_tokens,
messages=messages,
messages=effective_messages,
model=model,
metadata=metadata,
stop_sequences=stop_sequences,
stream=stream,
system=system,
system=effective_system,
temperature=temperature,
thinking=thinking,
tool_choice=tool_choice,
@ -336,7 +418,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
completion_response,
model=model,
tool_name_mapping=tool_name_mapping,
applied_edits=_polyfill_applied_edits,
polyfill_result=polyfill_result,
)
)
if transformed_stream is not None:
@ -346,7 +428,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params(
cast(ModelResponse, completion_response),
tool_name_mapping=tool_name_mapping,
applied_edits=_polyfill_applied_edits,
polyfill_result=polyfill_result,
)
if anthropic_response is not None:
return anthropic_response
@ -369,7 +451,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_p: Optional[float] = None,
output_format: Optional[Dict] = None,
_is_async: bool = False,
_polyfill_applied_edits: Optional[List[AppliedEdit]] = None,
**kwargs,
) -> Union[
AnthropicMessagesResponse,
@ -393,7 +474,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
top_k=top_k,
top_p=top_p,
output_format=output_format,
_polyfill_applied_edits=_polyfill_applied_edits,
**kwargs,
)
@ -426,7 +506,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
completion_response,
model=model,
tool_name_mapping=tool_name_mapping,
applied_edits=_polyfill_applied_edits,
)
)
if transformed_stream is not None:
@ -436,7 +515,6 @@ class LiteLLMMessagesToCompletionTransformationHandler:
anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params(
cast(ModelResponse, completion_response),
tool_name_mapping=tool_name_mapping,
applied_edits=_polyfill_applied_edits,
)
if anthropic_response is not None:
return anthropic_response

View file

@ -75,6 +75,9 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
from litellm.litellm_core_utils.prompt_templates.factory import (
THOUGHT_SIGNATURE_SEPARATOR,
)
from litellm.llms.anthropic.experimental_pass_through.context_management import (
PolyfillResult,
)
from litellm.types.llms.anthropic import (
ANTHROPIC_HOSTED_TOOLS,
AllAnthropicToolsValues,
@ -88,6 +91,7 @@ from litellm.types.llms.anthropic import (
AnthropicResponseContentBlockThinking,
AnthropicResponseContentBlockToolUse,
AppliedEdit,
CompactionBlock,
ContentBlockDelta,
ContentJsonBlockDelta,
ContentTextBlockDelta,
@ -97,6 +101,7 @@ from litellm.types.llms.anthropic import (
MessageBlockDelta,
MessageDelta,
UsageDelta,
UsageIteration,
)
from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
@ -197,7 +202,7 @@ class AnthropicAdapter:
self,
response: ModelResponse,
tool_name_mapping: Optional[Dict[str, str]] = None,
applied_edits: Optional[List[AppliedEdit]] = None,
polyfill_result: Optional[PolyfillResult] = None,
) -> Optional[AnthropicMessagesResponse]:
"""
Translate OpenAI response to Anthropic format.
@ -207,12 +212,12 @@ class AnthropicAdapter:
tool_name_mapping: Optional mapping of truncated tool names to original names.
Used to restore original names for tools that exceeded
OpenAI's 64-char limit.
applied_edits: Polyfill AppliedEdit list for response context_management.
polyfill_result: PolyfillResult from context_management polyfill.
"""
return LiteLLMAnthropicMessagesAdapter().translate_openai_response_to_anthropic(
response=response,
tool_name_mapping=tool_name_mapping,
applied_edits=applied_edits,
polyfill_result=polyfill_result,
)
def translate_completion_output_params_streaming(
@ -220,7 +225,7 @@ class AnthropicAdapter:
completion_stream: Any,
model: str,
tool_name_mapping: Optional[Dict[str, str]] = None,
applied_edits: Optional[List[AppliedEdit]] = None,
polyfill_result: Optional[PolyfillResult] = None,
) -> Union[AsyncIterator[bytes], None]:
"""
Translate OpenAI streaming response to Anthropic format.
@ -229,8 +234,9 @@ class AnthropicAdapter:
completion_stream: The OpenAI streaming response
model: The model name
tool_name_mapping: Optional mapping of truncated tool names to original names.
applied_edits: Polyfill AppliedEdit list on final message_delta.
polyfill_result: PolyfillResult from context_management polyfill.
"""
applied_edits = polyfill_result.applied_edits if polyfill_result else None
anthropic_wrapper = AnthropicStreamWrapper(
completion_stream=completion_stream,
model=model,
@ -1350,7 +1356,7 @@ class LiteLLMAnthropicMessagesAdapter:
self,
response: ModelResponse,
tool_name_mapping: Optional[Dict[str, str]] = None,
applied_edits: Optional[List[AppliedEdit]] = None,
polyfill_result: Optional[PolyfillResult] = None,
) -> AnthropicMessagesResponse:
"""
Translate OpenAI response to Anthropic format.
@ -1360,13 +1366,17 @@ class LiteLLMAnthropicMessagesAdapter:
tool_name_mapping: Optional mapping of truncated tool names to original names.
Used to restore original names for tools that exceeded
OpenAI's 64-char limit.
applied_edits: Polyfill AppliedEdit list for response context_management.
polyfill_result: PolyfillResult from context_management polyfill.
"""
## translate content block
anthropic_content = self._translate_openai_content_to_anthropic(
choices=response.choices, # type: ignore
tool_name_mapping=tool_name_mapping,
)
if polyfill_result is not None and polyfill_result.compaction_block is not None:
anthropic_content.insert(0, polyfill_result.compaction_block) # type: ignore[arg-type]
## extract finish reason
anthropic_finish_reason = self._translate_openai_finish_reason_to_anthropic(
openai_finish_reason=response.choices[0].finish_reason # type: ignore
@ -1395,6 +1405,14 @@ class LiteLLMAnthropicMessagesAdapter:
if cached_tokens > 0:
anthropic_usage["cache_read_input_tokens"] = cached_tokens
if polyfill_result is not None and polyfill_result.iterations_usage is not None:
message_iteration: UsageIteration = {
"type": "message",
"input_tokens": uncached_input_tokens,
"output_tokens": usage.completion_tokens or 0,
}
anthropic_usage["iterations"] = list(polyfill_result.iterations_usage) + [message_iteration] # type: ignore[typeddict-unknown-key]
translated_obj = AnthropicMessagesResponse(
id=response.id,
type="message",
@ -1406,6 +1424,7 @@ class LiteLLMAnthropicMessagesAdapter:
stop_reason=anthropic_finish_reason,
)
applied_edits = polyfill_result.applied_edits if polyfill_result else None
if applied_edits:
translated_obj["context_management"] = ContextManagementResponse(
applied_edits=list(applied_edits)

View file

@ -1,4 +1,11 @@
from .constants import CLEARED_TOOL_RESULT_PLACEHOLDER
from .dispatcher import apply_context_management
from .errors import AnthropicContextManagementError
from .result import PolyfillResult
__all__ = ["apply_context_management", "CLEARED_TOOL_RESULT_PLACEHOLDER"]
__all__ = [
"apply_context_management",
"AnthropicContextManagementError",
"CLEARED_TOOL_RESULT_PLACEHOLDER",
"PolyfillResult",
]

View file

@ -6,3 +6,28 @@ DEFAULT_INPUT_TOKENS_TRIGGER = 100_000
DEFAULT_KEEP_TOOL_USES = 3
CLEARED_TOOL_RESULT_PLACEHOLDER = "[Cleared by context management]"
# compact_20260112
COMPACT_EDIT_TYPE = "compact_20260112"
COMPACT_DEFAULT_TRIGGER_TOKENS = 150_000
COMPACT_MIN_TRIGGER_TOKENS = 50_000
COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model"
COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: "
# Default summarization prompt from the Anthropic spec.
COMPACT_DEFAULT_INSTRUCTIONS = (
"You have written a partial transcript for the initial task above. Please "
"write a summary of the transcript. The purpose of this summary is to "
"provide continuity so you can continue to make progress towards solving "
"the task in a future context, where the raw history above may not be "
"accessible and will be replaced with this summary. Write down anything "
"that would be helpful, including the state, next steps, learnings etc. "
"You must wrap your summary in a <summary></summary> block."
)
# Appended to the default prompt when ``tools`` are present and the caller
# did not supply custom ``instructions``. Matches the guidance in the
# Anthropic docs under "Compaction might fail when tools are defined".
COMPACT_NO_TOOL_CALLS_SUFFIX = (
" Do not call any tools while writing this summary; respond with text only."
)

View file

@ -1,56 +1,85 @@
"""Dispatch ``context_management`` edits to registered polyfill editors."""
from typing import Any, Callable, Dict, List, Optional, Tuple, Union
import inspect
from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple, Union, cast
from litellm._logging import verbose_logger
from litellm.types.llms.anthropic import AppliedEdit
from .constants import CLEAR_TOOL_USES_EDIT_TYPE
from .editors import apply_clear_tool_uses_20250919
from .constants import CLEAR_TOOL_USES_EDIT_TYPE, COMPACT_EDIT_TYPE
from .editors import apply_clear_tool_uses_20250919, apply_compact_20260112
from .result import PolyfillResult
EditorFn = Callable[..., Tuple[List[Dict[str, Any]], Optional[AppliedEdit]]]
EditorFn = Callable[..., Any]
_EDITOR_REGISTRY: Dict[str, EditorFn] = {
CLEAR_TOOL_USES_EDIT_TYPE: apply_clear_tool_uses_20250919,
COMPACT_EDIT_TYPE: apply_compact_20260112,
}
def apply_context_management(
def _normalize_spec(
spec: Union[Dict[str, Any], List[Dict[str, Any]], None],
) -> Optional[List[Dict[str, Any]]]:
"""Accept Anthropic-native dict form or OpenAI list form; return edits list."""
if isinstance(spec, list):
# Local import to avoid an import cycle at module load.
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
spec = AnthropicConfig.map_openai_context_management_to_anthropic(spec)
edits = spec.get("edits") if isinstance(spec, dict) else None
if not edits or not isinstance(edits, list):
return None
return [edit for edit in edits if isinstance(edit, dict)]
def _wrap_editor_return(raw: Any, *, fallback_system: Any) -> PolyfillResult:
"""Coerce an editor's native return shape into a ``PolyfillResult``.
v0 sync editors (e.g. ``clear_tool_uses_20250919``) return a 2-tuple
``(messages, Optional[AppliedEdit])``. The new async ``compact_20260112``
editor returns a ``PolyfillResult`` directly.
"""
if isinstance(raw, PolyfillResult):
return raw
# Legacy 2-tuple return — sync editors don't mutate ``system``, so
# carry the caller's value forward.
messages, applied = cast(Tuple[List[Dict[str, Any]], Any], raw)
return PolyfillResult(
messages=messages,
system=fallback_system,
applied_edits=[applied] if applied is not None else [],
)
async def apply_context_management(
*,
model: str,
messages: List[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]],
system: Any,
context_management_spec: Union[Dict[str, Any], List[Dict[str, Any]], None],
) -> Tuple[List[Dict[str, Any]], List[AppliedEdit]]:
"""Run edits in order; return (messages, applied_edits that fired)."""
# Accept both Anthropic-native dict form and OpenAI list form. The other
# provider paths normalize via ``map_openai_context_management_to_anthropic``
# before dispatching; do the same here so the polyfill path doesn't silently
# no-op on list input.
if isinstance(context_management_spec, list):
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
metadata: Optional[Dict[str, Any]] = None,
llm_router: Any = None,
) -> PolyfillResult:
"""Run edits in order; return a single ``PolyfillResult``.
context_management_spec = (
AnthropicConfig.map_openai_context_management_to_anthropic(
context_management_spec
)
)
The dispatcher is async so async editors (``compact_20260112``) can
``await`` the configured summarization model. Sync editors are called
inline — ``inspect.iscoroutinefunction`` decides how each editor is
invoked.
"""
edits = _normalize_spec(context_management_spec)
if not edits:
return PolyfillResult(messages=messages, system=system, applied_edits=[])
edits = (
context_management_spec.get("edits")
if isinstance(context_management_spec, dict)
else None
)
if not edits or not isinstance(edits, list):
return messages, []
applied_edits: List[AppliedEdit] = []
current_messages = messages
current_system = system
aggregated_applied: List[Dict[str, Any]] = []
aggregated_compaction_block = None
aggregated_iterations_usage = None
for edit_spec in edits:
if not isinstance(edit_spec, dict):
continue
edit_type = edit_spec.get("type")
editor = _EDITOR_REGISTRY.get(edit_type) if isinstance(edit_type, str) else None
if editor is None:
@ -60,14 +89,36 @@ def apply_context_management(
)
continue
current_messages, applied = editor(
model=model,
messages=current_messages,
tools=tools,
system=system,
edit_spec=edit_spec,
)
if applied is not None:
applied_edits.append(applied)
kwargs: Dict[str, Any] = {
"model": model,
"messages": current_messages,
"tools": tools,
"system": current_system,
"edit_spec": edit_spec,
}
# Only async editors accept these — passing them to sync v0 editors
# would break their signature.
if inspect.iscoroutinefunction(editor):
kwargs["metadata"] = metadata
kwargs["llm_router"] = llm_router
raw_result = await cast(Callable[..., Awaitable[Any]], editor)(**kwargs)
else:
raw_result = editor(**kwargs)
return current_messages, applied_edits
result = _wrap_editor_return(raw_result, fallback_system=current_system)
current_messages = result.messages
current_system = result.system
aggregated_applied.extend(result.applied_edits)
if result.compaction_block is not None:
aggregated_compaction_block = result.compaction_block
if result.iterations_usage is not None:
aggregated_iterations_usage = result.iterations_usage
return PolyfillResult(
messages=current_messages,
system=current_system,
applied_edits=cast(List[Any], aggregated_applied),
compaction_block=aggregated_compaction_block,
iterations_usage=aggregated_iterations_usage,
)

View file

@ -1,3 +1,4 @@
from .clear_tool_uses import apply_clear_tool_uses_20250919
from .compact import apply_compact_20260112
__all__ = ["apply_clear_tool_uses_20250919"]
__all__ = ["apply_clear_tool_uses_20250919", "apply_compact_20260112"]

View file

@ -0,0 +1,494 @@
"""``compact_20260112`` polyfill (server-side context compaction).
Mirrors Anthropic's native ``compact_20260112`` for non-Anthropic providers:
- Scans the message history for an existing ``compaction`` block; everything
before it is dropped (slice).
- If still over the configured trigger, calls a separately-configured
summarization model and synthesizes a fresh ``compaction`` block.
- The summary is injected as a system-message prefix on the downstream call
(the user/assistant log carries no ``compaction`` block downstream).
- The synthesized ``compaction`` block is returned via ``PolyfillResult`` so
the response adapter can prepend it to the response ``content`` array.
"""
import re
from typing import Any, Dict, List, Optional, Tuple, Union, cast
import litellm
from litellm._logging import verbose_logger
from litellm.types.llms.anthropic import (
AppliedEdit,
CompactionBlock,
UsageIteration,
)
from ..constants import (
COMPACT_DEFAULT_INSTRUCTIONS,
COMPACT_DEFAULT_TRIGGER_TOKENS,
COMPACT_EDIT_TYPE,
COMPACT_MIN_TRIGGER_TOKENS,
COMPACT_NO_TOOL_CALLS_SUFFIX,
COMPACT_SUMMARY_MODEL_SETTING_KEY,
COMPACT_SUMMARY_SYSTEM_PREFIX,
)
from ..errors import AnthropicContextManagementError
from ..result import PolyfillResult
# Auth metadata fields propagated from the parent request to the summary call
# so the summary's spend is attributed to the same team/key. The list mirrors
# the fields populated by
# ``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata``.
_PROPAGATED_METADATA_KEYS = (
"user_api_key",
"user_api_key_alias",
"user_api_key_team_id",
"user_api_key_team_alias",
"user_api_key_user_id",
"user_api_key_user_email",
"user_api_key_org_id",
"litellm_call_id",
"litellm_parent_otel_span",
)
_SUMMARY_TAG_RE = re.compile(r"<summary>(.*?)</summary>", re.IGNORECASE | re.DOTALL)
def _read_summary_model_setting() -> Optional[str]:
"""Look up the configured summarization model from proxy general_settings."""
try:
from litellm.proxy.proxy_server import general_settings
except Exception:
return None
value = general_settings.get(COMPACT_SUMMARY_MODEL_SETTING_KEY)
return value if isinstance(value, str) and value else None
def _find_latest_compaction_index(
messages: List[Dict[str, Any]],
) -> Tuple[Optional[int], Optional[int]]:
"""Return (message_index, block_index) of the most recent compaction block.
``None, None`` if no compaction block is present. Iterates from the end so
only the latest one is considered.
"""
for msg_idx in range(len(messages) - 1, -1, -1):
content = messages[msg_idx].get("content")
if not isinstance(content, list):
continue
for blk_idx in range(len(content) - 1, -1, -1):
block = content[blk_idx]
if isinstance(block, dict) and block.get("type") == "compaction":
return msg_idx, blk_idx
return None, None
def _slice_around_compaction_block(
messages: List[Dict[str, Any]],
) -> Tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]]:
"""Apply Anthropic's "drop everything before the compaction block" rule.
Returns ``(sliced_messages_with_compaction_block, compaction_block_dict)``
if a block was found, else ``(original_messages, None)``. The sliced result
keeps the compaction block in the assistant turn that originally carried
it (in practice it's the only block in that turn) so callers can still
extract the summary text from it.
"""
msg_idx, blk_idx = _find_latest_compaction_index(messages)
if msg_idx is None or blk_idx is None:
return messages, None
original_msg = messages[msg_idx]
original_content = original_msg["content"]
compaction_block = cast(Dict[str, Any], original_content[blk_idx])
# Per Anthropic's contract everything before the compaction block is
# dropped, including earlier blocks within the same assistant message.
sliced_content = list(original_content[blk_idx:])
sliced_first_msg = {**original_msg, "content": sliced_content}
sliced_messages: List[Dict[str, Any]] = [sliced_first_msg]
sliced_messages.extend(messages[msg_idx + 1 :])
return sliced_messages, compaction_block
def _strip_compaction_blocks(
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
"""Drop any ``compaction`` content blocks from messages.
Used to build the downstream-bound message list — the adapter has no
concept of a compaction block, so it must not see one.
"""
cleaned: List[Dict[str, Any]] = []
for msg in messages:
content = msg.get("content")
if not isinstance(content, list):
cleaned.append(msg)
continue
filtered = [
block
for block in content
if not (isinstance(block, dict) and block.get("type") == "compaction")
]
if not filtered:
# The compaction block was the only content; drop the whole turn.
continue
cleaned.append({**msg, "content": filtered})
return cleaned
def _augment_system_with_summary(
system: Optional[Union[str, List[Dict[str, Any]]]],
summary_text: str,
) -> Union[str, List[Dict[str, Any]]]:
"""Prepend a "Previous conversation summary: ..." block to ``system``."""
prefix = f"{COMPACT_SUMMARY_SYSTEM_PREFIX}{summary_text}\n\n"
if system is None:
return prefix.rstrip()
if isinstance(system, str):
return f"{prefix}{system}"
# List of content blocks: prepend the prefix to the first text block,
# otherwise insert a new text block at the head.
for idx, block in enumerate(system):
if isinstance(block, dict) and block.get("type") == "text":
existing = block.get("text", "") or ""
new_block = {**block, "text": f"{prefix}{existing}"}
return [*system[:idx], new_block, *system[idx + 1 :]]
return [{"type": "text", "text": prefix.rstrip()}, *system]
def _resolve_trigger_tokens(edit_spec: Dict[str, Any]) -> Tuple[int, List[str]]:
"""Validate and resolve ``trigger.value``.
Raises ``AnthropicContextManagementError`` if the explicitly-supplied value
is below the 50k minimum. Unknown ``trigger.type`` values fall back to
``input_tokens`` with a warning.
"""
warnings: List[str] = []
trigger = edit_spec.get("trigger") or {}
if not isinstance(trigger, dict):
warnings.append("trigger_not_a_dict_using_default")
return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
trigger_type = trigger.get("type", "input_tokens")
if trigger_type != "input_tokens":
warnings.append(f"unsupported_trigger_type_{trigger_type}_using_input_tokens")
value = trigger.get("value")
if value is None:
return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
if not isinstance(value, int):
warnings.append("trigger_value_not_int_using_default")
return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings
if value < COMPACT_MIN_TRIGGER_TOKENS:
raise AnthropicContextManagementError(
status_code=400,
message=(
f"context_management.compact_20260112.trigger.value must be at "
f"least {COMPACT_MIN_TRIGGER_TOKENS} tokens"
),
)
return value, warnings
def _build_summary_prompt(
edit_spec: Dict[str, Any], tools: Optional[List[Dict[str, Any]]]
) -> str:
custom = edit_spec.get("instructions")
if isinstance(custom, str) and custom.strip():
return custom
prompt = COMPACT_DEFAULT_INSTRUCTIONS
if tools:
prompt = f"{prompt}{COMPACT_NO_TOOL_CALLS_SUFFIX}"
return prompt
def _propagate_metadata(parent_metadata: Optional[Dict[str, Any]]) -> Dict[str, Any]:
if not parent_metadata:
return {}
propagated: Dict[str, Any] = {}
for key in _PROPAGATED_METADATA_KEYS:
if key in parent_metadata:
propagated[key] = parent_metadata[key]
return propagated
def _count_effective_tokens(
model: str,
effective_messages: List[Dict[str, Any]],
compaction_block: Optional[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]],
) -> int:
"""Token-count the conversation as it will appear downstream.
The compaction block (if any) becomes a system prefix on the downstream
call, so its content still counts even though it isn't in ``messages``.
"""
# Local import to avoid pulling the adapter at module load time.
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
messages_without_compaction = _strip_compaction_blocks(effective_messages)
try:
openai_shape = (
LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=cast(Any, messages_without_compaction)
)
)
except Exception as e:
verbose_logger.debug(
"compact_20260112: anthropic→openai translation failed during token "
"count, falling back to raw messages: %s",
e,
)
openai_shape = cast(Any, messages_without_compaction)
total = litellm.token_counter(
model=model,
messages=cast(Any, openai_shape),
tools=cast(Any, tools),
)
if compaction_block is not None:
content = compaction_block.get("content") or ""
if content:
total += litellm.token_counter(model=model, text=content)
return total
def _extract_summary_text(raw: Optional[str]) -> Optional[str]:
if not raw:
return None
match = _SUMMARY_TAG_RE.search(raw)
if match is None:
return None
summary = match.group(1).strip()
return summary or None
def _build_summary_messages(
effective_messages: List[Dict[str, Any]],
prompt: str,
) -> List[Dict[str, Any]]:
"""Build the OpenAI-shape message list for the summary call.
The conversation history is translated to OpenAI shape; the
summarization prompt is appended as a final user turn.
"""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
stripped = _strip_compaction_blocks(effective_messages)
try:
openai_messages = (
LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=cast(Any, stripped)
)
)
except Exception as e:
verbose_logger.warning(
"compact_20260112: anthropic→openai translation failed when "
"building summary call; falling back to raw shape: %s",
e,
)
openai_messages = cast(Any, stripped)
return [*openai_messages, {"role": "user", "content": prompt}]
async def _call_summary_model(
*,
summary_model: str,
summary_messages: List[Dict[str, Any]],
metadata: Dict[str, Any],
llm_router: Any,
) -> Any:
"""Invoke the configured summary model.
Prefers ``llm_router.acompletion`` so the model alias resolves against the
proxy's ``model_list``; falls back to ``litellm.acompletion`` if no router
is available (e.g. SDK usage outside the proxy).
"""
call_kwargs: Dict[str, Any] = {
"model": summary_model,
"messages": summary_messages,
"metadata": metadata,
}
if llm_router is not None and hasattr(llm_router, "acompletion"):
return await llm_router.acompletion(**call_kwargs)
return await litellm.acompletion(**call_kwargs)
def _extract_response_text(response: Any) -> Optional[str]:
try:
choice = response.choices[0]
message = choice.message
content = getattr(message, "content", None)
if isinstance(content, str):
return content
# Some providers return a list of content parts.
if isinstance(content, list):
text_parts = [
part.get("text", "")
for part in content
if isinstance(part, dict) and part.get("type") == "text"
]
return "".join(text_parts) or None
except (AttributeError, IndexError, KeyError):
return None
return None
def _extract_usage(response: Any) -> Tuple[int, int]:
usage = getattr(response, "usage", None)
if usage is None:
return 0, 0
return (
int(getattr(usage, "prompt_tokens", 0) or 0),
int(getattr(usage, "completion_tokens", 0) or 0),
)
async def apply_compact_20260112(
*,
model: str,
messages: List[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]],
system: Optional[Union[str, List[Dict[str, Any]]]],
edit_spec: Dict[str, Any],
metadata: Optional[Dict[str, Any]] = None,
llm_router: Any = None,
) -> PolyfillResult:
"""Apply ``compact_20260112``; return a ``PolyfillResult``.
See module docstring for the algorithm. Errors are best-effort: when the
summary call fails or the response is malformed, the editor returns the
pre-summary state (with ``applied_edits[0].error`` populated) so the
original request still proceeds.
"""
# Validation runs first. Raising AnthropicContextManagementError here is
# the only path on which the polyfill aborts the request.
trigger_tokens, warnings = _resolve_trigger_tokens(edit_spec)
if edit_spec.get("pause_after_compaction"):
warnings.append("pause_after_compaction_ignored")
applied: AppliedEdit = {"type": COMPACT_EDIT_TYPE}
if warnings:
applied["warnings"] = warnings
# Opt-in gate: no summary model configured → no-op.
summary_model = _read_summary_model_setting()
if summary_model is None:
applied["error"] = "summary_model_not_configured"
return PolyfillResult(
messages=messages,
system=system,
applied_edits=[applied],
)
# Phase A: slice around any existing compaction block.
effective_messages, prior_compaction_block = _slice_around_compaction_block(
messages
)
prior_summary_text = (
prior_compaction_block.get("content") if prior_compaction_block else None
)
augmented_system: Union[str, List[Dict[str, Any]], None] = system
if isinstance(prior_summary_text, str) and prior_summary_text:
augmented_system = _augment_system_with_summary(system, prior_summary_text)
downstream_messages = _strip_compaction_blocks(effective_messages)
# Phase B: threshold check.
try:
current_tokens = _count_effective_tokens(
model=model,
effective_messages=effective_messages,
compaction_block=prior_compaction_block,
tools=tools,
)
except Exception as e:
verbose_logger.warning(
"compact_20260112: token_counter failed; assuming under threshold: %s", e
)
current_tokens = 0
verbose_logger.debug(
"compact_20260112: current_tokens=%s trigger=%s", current_tokens, trigger_tokens
)
if current_tokens <= trigger_tokens:
# Slice-only path. If Phase A fired we still slice + system-prefix.
return PolyfillResult(
messages=downstream_messages,
system=augmented_system,
applied_edits=[applied],
)
# Phase C: summarize.
prompt = _build_summary_prompt(edit_spec, tools)
summary_messages = _build_summary_messages(effective_messages, prompt)
propagated_metadata = _propagate_metadata(metadata)
try:
response = await _call_summary_model(
summary_model=summary_model,
summary_messages=summary_messages,
metadata=propagated_metadata,
llm_router=llm_router,
)
except Exception as e:
verbose_logger.warning("compact_20260112: summary call failed: %s", e)
applied["error"] = "summary_call_failed"
return PolyfillResult(
messages=downstream_messages,
system=augmented_system,
applied_edits=[applied],
)
summary_text = _extract_summary_text(_extract_response_text(response))
if summary_text is None:
applied["error"] = "summary_extraction_failed"
return PolyfillResult(
messages=downstream_messages,
system=augmented_system,
applied_edits=[applied],
)
summary_input_tokens, summary_output_tokens = _extract_usage(response)
applied["summary_input_tokens"] = summary_input_tokens
applied["summary_output_tokens"] = summary_output_tokens
compaction_block: CompactionBlock = {
"type": "compaction",
"content": summary_text,
}
iterations_usage: List[UsageIteration] = [
{
"type": "compaction",
"input_tokens": summary_input_tokens,
"output_tokens": summary_output_tokens,
}
]
# Per Anthropic's contract, everything before the compaction block is
# dropped. Phase D: the user/assistant log goes empty; the summary lives
# on the system message instead. Anthropic requires a non-empty messages
# array, so keep the most recent original user turn so the model has the
# question to answer.
summarized_system = _augment_system_with_summary(system, summary_text)
downstream_messages_after_summary: List[Dict[str, Any]] = []
for msg in reversed(messages):
if msg.get("role") == "user":
downstream_messages_after_summary = [msg]
break
return PolyfillResult(
messages=downstream_messages_after_summary,
system=summarized_system,
applied_edits=[applied],
compaction_block=compaction_block,
iterations_usage=iterations_usage,
)

View file

@ -0,0 +1,14 @@
"""Exceptions raised by the context_management polyfill."""
class AnthropicContextManagementError(Exception):
"""Validation error from the polyfill, surfaced as an Anthropic-format 4xx.
The `/v1/messages` endpoint catches this in its exception handler and
emits an Anthropic-shaped error body instead of the default OpenAI shape.
"""
def __init__(self, *, status_code: int, message: str) -> None:
super().__init__(message)
self.status_code = status_code
self.message = message

View file

@ -0,0 +1,24 @@
"""``PolyfillResult`` — the shape returned by the context-management dispatcher.
Threaded from the dispatcher through ``async_anthropic_messages_handler`` into
the adapter so it can prepend the ``compaction`` block to the response and
attach ``iterations`` to ``usage``.
"""
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional, Union
from litellm.types.llms.anthropic import (
AppliedEdit,
CompactionBlock,
UsageIteration,
)
@dataclass
class PolyfillResult:
messages: List[Dict[str, Any]]
system: Optional[Union[str, List[Dict[str, Any]]]]
applied_edits: List[AppliedEdit] = field(default_factory=list)
compaction_block: Optional[CompactionBlock] = None
iterations_usage: Optional[List[UsageIteration]] = None

View file

@ -459,40 +459,12 @@ def anthropic_messages_handler(
**_shared_kwargs
)
# In-gateway context_management polyfill on the chat-completions adapter
# (native on Anthropic/Responses paths). Skipped when drop_params is on.
context_management_spec = _shared_kwargs.pop("context_management", None)
_request_drop_params = _shared_kwargs.get("drop_params")
_drop_params = (
_request_drop_params
if _request_drop_params is not None
else litellm.drop_params
)
polyfill_applied_edits: Optional[List[AppliedEdit]] = None
if context_management_spec and not _drop_params:
from litellm.llms.anthropic.experimental_pass_through.context_management import (
apply_context_management,
)
try:
edited_messages, polyfill_applied_edits = apply_context_management(
model=model,
messages=_shared_kwargs["messages"],
tools=_shared_kwargs.get("tools"),
system=_shared_kwargs.get("system"),
context_management_spec=context_management_spec,
)
_shared_kwargs["messages"] = edited_messages
except Exception as e:
verbose_logger.exception(
"context_management polyfill: skipping edits due to error: %s",
e,
)
polyfill_applied_edits = None
# The in-gateway context_management polyfill runs inside
# ``async_anthropic_messages_handler`` so it can ``await`` the
# summarization model for ``compact_20260112``. ``context_management``
# is passed through as a regular kwarg.
return (
LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler(
_polyfill_applied_edits=polyfill_applied_edits,
**_shared_kwargs,
)
)

View file

@ -3,10 +3,14 @@ Unified /v1/messages endpoint - (Anthropic Spec)
"""
from fastapi import APIRouter, Depends, HTTPException, Request, Response
from fastapi.responses import JSONResponse
from litellm._logging import verbose_proxy_logger
from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping
from litellm.integrations.custom_guardrail import ModifyResponseException
from litellm.llms.anthropic.experimental_pass_through.context_management import (
AnthropicContextManagementError,
)
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_request_processing import (
@ -114,6 +118,13 @@ async def anthropic_response( # noqa: PLR0915
)
return _anthropic_response
except AnthropicContextManagementError as e:
body = AnthropicExceptionMapping.transform_to_anthropic_error(
status_code=e.status_code,
raw_message=e.message,
request_id=request.headers.get("x-request-id"),
)
return JSONResponse(status_code=e.status_code, content=body)
except Exception as e:
await proxy_logging_obj.post_call_failure_hook(
user_api_key_dict=user_api_key_dict, original_exception=e, request_data=data

View file

@ -521,6 +521,11 @@ class AppliedEdit(TypedDict, total=False):
cleared_input_tokens: int
cleared_tool_uses: int
cleared_thinking_turns: int
# compact_20260112 fields
summary_input_tokens: int
summary_output_tokens: int
error: str
warnings: List[str]
class ContextManagementResponse(TypedDict, total=False):
@ -529,6 +534,21 @@ class ContextManagementResponse(TypedDict, total=False):
applied_edits: List[AppliedEdit]
class CompactionBlock(TypedDict, total=False):
"""Synthesized ``compaction`` content block (compact_20260112)."""
type: Literal["compaction"]
content: Optional[str]
class UsageIteration(TypedDict, total=False):
"""One sampling iteration's token usage (compact_20260112)."""
type: Literal["compaction", "message"]
input_tokens: int
output_tokens: int
class MessageBlockDelta(TypedDict):
"""
Anthropic

View file

@ -2472,3 +2472,172 @@ def test_translate_anthropic_tool_choice_none():
result = adapter.translate_anthropic_tool_choice_to_openai({"type": "none"})
assert result == "none"
# ---------------------------------------------------------------------------
# PolyfillResult integration tests
# ---------------------------------------------------------------------------
def _make_simple_openai_response(
text: str = "Hello", prompt_tokens: int = 10, completion_tokens: int = 5
) -> ModelResponse:
return ModelResponse(
id="resp_polyfill_test",
model="gpt-4o",
choices=[
Choices(
finish_reason="stop",
message=Message(role="assistant", content=text),
)
],
usage=Usage(prompt_tokens=prompt_tokens, completion_tokens=completion_tokens),
)
def test_translate_openai_response_to_anthropic_with_polyfill_compaction_block():
"""compaction_block from PolyfillResult must be prepended to content at index 0."""
from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
PolyfillResult,
)
compaction_block = {"type": "compaction", "content": "Summary of prior turns."}
polyfill = PolyfillResult(
messages=[],
system=None,
applied_edits=[{"type": "compact_20260112"}],
compaction_block=compaction_block,
iterations_usage=None,
)
response = _make_simple_openai_response(text="Hello after compaction.")
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_openai_response_to_anthropic(
response=response, polyfill_result=polyfill
)
content = result.get("content")
assert content is not None
assert content[0]["type"] == "compaction"
assert content[0]["content"] == "Summary of prior turns."
assert content[1]["type"] == "text"
assert content[1]["text"] == "Hello after compaction."
# applied_edits must surface on context_management
cm = result.get("context_management")
assert cm is not None
assert cm["applied_edits"][0]["type"] == "compact_20260112"
def test_translate_openai_response_to_anthropic_with_polyfill_iterations_usage():
"""iterations_usage from PolyfillResult must produce usage['iterations'] with a message entry."""
from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
PolyfillResult,
)
polyfill = PolyfillResult(
messages=[],
system=None,
applied_edits=[{"type": "compact_20260112"}],
compaction_block=None,
iterations_usage=[
{"type": "compaction", "input_tokens": 200, "output_tokens": 50},
],
)
response = _make_simple_openai_response(prompt_tokens=100, completion_tokens=30)
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_openai_response_to_anthropic(
response=response, polyfill_result=polyfill
)
usage = result.get("usage")
assert usage is not None
iterations = usage.get("iterations")
assert iterations is not None
assert len(iterations) == 2
assert iterations[0] == {
"type": "compaction",
"input_tokens": 200,
"output_tokens": 50,
}
assert iterations[1]["type"] == "message"
assert iterations[1]["input_tokens"] == 100
assert iterations[1]["output_tokens"] == 30
# Top-level tokens must still reflect the message iteration
assert usage["input_tokens"] == 100
assert usage["output_tokens"] == 30
def test_translate_openai_response_to_anthropic_no_polyfill_no_change():
"""Without a PolyfillResult the response must be unchanged (no compaction, no iterations)."""
response = _make_simple_openai_response()
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_openai_response_to_anthropic(response=response)
content = result.get("content")
assert content is not None
assert content[0]["type"] == "text"
usage = result.get("usage")
assert usage is not None
assert "iterations" not in usage
def test_translate_openai_response_to_anthropic_with_polyfill_both_compaction_and_iterations():
"""Full summary path: compaction_block and iterations_usage both present simultaneously."""
from litellm.llms.anthropic.experimental_pass_through.context_management.result import (
PolyfillResult,
)
compaction_block = {
"type": "compaction",
"content": "Summary of a long conversation.",
}
polyfill = PolyfillResult(
messages=[],
system=None,
applied_edits=[{"type": "compact_20260112"}],
compaction_block=compaction_block,
iterations_usage=[
{"type": "compaction", "input_tokens": 300, "output_tokens": 75},
],
)
response = _make_simple_openai_response(
text="After compaction.", prompt_tokens=120, completion_tokens=40
)
adapter = LiteLLMAnthropicMessagesAdapter()
result = adapter.translate_openai_response_to_anthropic(
response=response, polyfill_result=polyfill
)
# compaction block must come first
content = result.get("content")
assert content is not None
assert content[0]["type"] == "compaction"
assert content[0]["content"] == "Summary of a long conversation."
assert content[1]["type"] == "text"
assert content[1]["text"] == "After compaction."
# iterations: compaction entry + message entry
usage = result.get("usage")
assert usage is not None
iterations = usage.get("iterations")
assert iterations is not None
assert len(iterations) == 2
assert iterations[0] == {
"type": "compaction",
"input_tokens": 300,
"output_tokens": 75,
}
assert iterations[1]["type"] == "message"
assert iterations[1]["input_tokens"] == 120
assert iterations[1]["output_tokens"] == 40
# top-level tokens match the message iteration
assert usage["input_tokens"] == 120
assert usage["output_tokens"] == 40
# context_management applied_edits must surface
cm = result.get("context_management")
assert cm is not None
assert cm["applied_edits"][0]["type"] == "compact_20260112"

View file

@ -0,0 +1,862 @@
"""
Unit tests for the compact_20260112 polyfill editor.
Coverage:
- trigger.value < 50k → AnthropicContextManagementError(400)
- opt-in gate (no summary model) → summary_model_not_configured
- slice-only path (existing compaction block, under threshold)
- full summary path (over threshold, summary fires)
- summary call raises → summary_call_failed
- summary response missing <summary> tags → summary_extraction_failed
- pause_after_compaction: true → pause_after_compaction_ignored warning, proceeds
- custom instructions → default prompt is not used even when tools present
"""
from typing import Any, Dict, List
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from litellm.llms.anthropic.experimental_pass_through.context_management import (
AnthropicContextManagementError,
apply_context_management,
)
from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import (
_augment_system_with_summary,
_extract_summary_text,
_slice_around_compaction_block,
_strip_compaction_blocks,
apply_compact_20260112,
)
MODEL = "openai/gpt-4o"
_EDIT_SPEC_DEFAULT: Dict[str, Any] = {"type": "compact_20260112"}
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _simple_messages() -> List[Dict[str, Any]]:
return [
{"role": "user", "content": "Hello"},
{"role": "assistant", "content": [{"type": "text", "text": "Hi there"}]},
{"role": "user", "content": "What is 2+2?"},
]
def _messages_with_compaction(summary: str = "prev summary") -> List[Dict[str, Any]]:
"""History that already has a compaction block in an assistant turn."""
return [
{"role": "user", "content": "older question"},
{
"role": "assistant",
"content": [{"type": "compaction", "content": summary}],
},
{"role": "user", "content": "newer question"},
{"role": "assistant", "content": [{"type": "text", "text": "newer reply"}]},
{"role": "user", "content": "latest question"},
]
def _make_mock_response(
content: str,
prompt_tokens: int = 50,
completion_tokens: int = 100,
) -> MagicMock:
response = MagicMock()
choice = MagicMock()
message = MagicMock()
message.content = content
choice.message = message
response.choices = [choice]
usage = MagicMock()
usage.prompt_tokens = prompt_tokens
usage.completion_tokens = completion_tokens
response.usage = usage
return response
# ---------------------------------------------------------------------------
# Unit: helper functions
# ---------------------------------------------------------------------------
def test_slice_around_compaction_block_found():
messages = _messages_with_compaction("my summary")
sliced, block = _slice_around_compaction_block(messages)
assert block is not None
assert block["type"] == "compaction"
assert block["content"] == "my summary"
# Sliced list starts at the assistant turn containing the compaction block
assert sliced[0]["role"] == "assistant"
assert len(sliced) == 4 # assistant(compaction), user, assistant, user
def test_slice_around_compaction_block_not_found():
messages = _simple_messages()
sliced, block = _slice_around_compaction_block(messages)
assert block is None
assert sliced is messages # same object, no copy
def test_strip_compaction_blocks_removes_block():
messages = [
{
"role": "assistant",
"content": [
{"type": "compaction", "content": "summary"},
{"type": "text", "text": "hello"},
],
}
]
stripped = _strip_compaction_blocks(messages)
assert len(stripped) == 1
content = stripped[0]["content"]
assert all(b["type"] != "compaction" for b in content)
assert len(content) == 1
assert content[0]["type"] == "text"
def test_strip_compaction_blocks_drops_compaction_only_turn():
messages = [
{"role": "user", "content": "hi"},
{
"role": "assistant",
"content": [{"type": "compaction", "content": "summary"}],
},
{"role": "user", "content": "bye"},
]
stripped = _strip_compaction_blocks(messages)
assert len(stripped) == 2
assert stripped[0]["role"] == "user"
assert stripped[1]["role"] == "user"
def test_augment_system_with_summary_none_system():
result = _augment_system_with_summary(None, "my summary")
assert isinstance(result, str)
assert "my summary" in result
def test_augment_system_with_summary_string_system():
result = _augment_system_with_summary("You are helpful.", "my summary")
assert isinstance(result, str)
assert result.startswith("Previous conversation summary:")
assert "my summary" in result
assert "You are helpful." in result
def test_augment_system_with_summary_list_system():
system = [{"type": "text", "text": "existing system"}]
result = _augment_system_with_summary(system, "my summary")
assert isinstance(result, list)
assert result[0]["type"] == "text"
text = result[0]["text"]
assert "my summary" in text
assert "existing system" in text
def test_extract_summary_text_found():
raw = "Here is the summary:\n<summary>Key points from chat</summary>\nDone."
assert _extract_summary_text(raw) == "Key points from chat"
def test_extract_summary_text_missing_tags():
assert _extract_summary_text("No tags here") is None
def test_extract_summary_text_none():
assert _extract_summary_text(None) is None
def test_extract_summary_text_case_insensitive():
raw = "<SUMMARY>uppercase tags</SUMMARY>"
assert _extract_summary_text(raw) == "uppercase tags"
# ---------------------------------------------------------------------------
# Editor: validation
# ---------------------------------------------------------------------------
async def test_trigger_below_minimum_raises():
with pytest.raises(AnthropicContextManagementError) as exc_info:
await apply_compact_20260112(
model=MODEL,
messages=_simple_messages(),
tools=None,
system=None,
edit_spec={
"type": "compact_20260112",
"trigger": {"type": "input_tokens", "value": 10_000},
},
)
assert exc_info.value.status_code == 400
assert "50000" in exc_info.value.message
async def test_trigger_at_minimum_does_not_raise():
"""Exactly 50 000 is allowed — only strictly less than 50k is rejected."""
with patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value=None,
):
result = await apply_compact_20260112(
model=MODEL,
messages=_simple_messages(),
tools=None,
system=None,
edit_spec={
"type": "compact_20260112",
"trigger": {"type": "input_tokens", "value": 50_000},
},
)
# Reached opt-in gate (no summary model); no error raised from trigger check
assert result.applied_edits[0]["error"] == "summary_model_not_configured"
# ---------------------------------------------------------------------------
# Editor: opt-in gate
# ---------------------------------------------------------------------------
async def test_opt_in_gating_no_summary_model_configured():
messages = _simple_messages()
with patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value=None,
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system="system prompt",
edit_spec=_EDIT_SPEC_DEFAULT,
)
assert result.applied_edits[0]["error"] == "summary_model_not_configured"
assert result.messages == messages
assert result.system == "system prompt"
assert result.compaction_block is None
assert result.iterations_usage is None
# ---------------------------------------------------------------------------
# Editor: slice-only path
# ---------------------------------------------------------------------------
async def test_slice_only_path_with_existing_compaction_block():
"""Phase A slices; Phase B token count is below threshold; no summary call."""
messages = _messages_with_compaction("prior summary text")
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=500), # well under threshold
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
# System should have the prior summary prefixed
assert result.system is not None
assert "prior summary text" in str(result.system)
# No new compaction block; no iterations_usage
assert result.compaction_block is None
assert result.iterations_usage is None
# No compaction block in downstream messages
for msg in result.messages:
content = msg.get("content")
if isinstance(content, list):
for block in content:
assert block.get("type") != "compaction"
async def test_slice_only_no_compaction_block_under_threshold():
"""No prior compaction block, and token count is below threshold — pure pass-through."""
messages = _simple_messages()
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=500),
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
assert result.messages == messages
assert result.compaction_block is None
assert result.iterations_usage is None
assert not result.applied_edits[0].get("error")
# ---------------------------------------------------------------------------
# Editor: full summary path
# ---------------------------------------------------------------------------
async def test_full_summary_path():
"""Over threshold: summary call fires, compaction_block and iterations_usage returned."""
messages = _simple_messages()
mock_response = _make_mock_response(
"<summary>Condensed history</summary>", prompt_tokens=200, completion_tokens=50
)
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000), # over 150k threshold
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
new_callable=AsyncMock,
return_value=mock_response,
),
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
assert result.compaction_block is not None
assert result.compaction_block["type"] == "compaction"
assert result.compaction_block["content"] == "Condensed history"
assert result.iterations_usage is not None
assert len(result.iterations_usage) == 1
assert result.iterations_usage[0]["type"] == "compaction"
assert result.iterations_usage[0]["input_tokens"] == 200
assert result.iterations_usage[0]["output_tokens"] == 50
# System must have summary prefixed
assert "Condensed history" in str(result.system)
# applied_edits should have usage fields
edit = result.applied_edits[0]
assert edit["type"] == "compact_20260112"
assert edit.get("summary_input_tokens") == 200
assert edit.get("summary_output_tokens") == 50
# Downstream messages must not contain a compaction block
for msg in result.messages:
content = msg.get("content")
if isinstance(content, list):
for block in content:
assert block.get("type") != "compaction"
async def test_full_summary_path_uses_router_when_available():
"""When llm_router is provided, its acompletion method is called instead of litellm."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>Router summary</summary>")
mock_router = MagicMock()
mock_router.acompletion = AsyncMock(return_value=mock_response)
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="my-summary-model",
),
patch("litellm.token_counter", return_value=200_000),
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
llm_router=mock_router,
)
mock_router.acompletion.assert_called_once()
call_kwargs = mock_router.acompletion.call_args.kwargs
assert call_kwargs["model"] == "my-summary-model"
assert result.compaction_block is not None
assert result.compaction_block["content"] == "Router summary"
async def test_metadata_propagated_to_summary_call():
"""Auth metadata from the parent request is forwarded to the summary call."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>Summary</summary>")
parent_metadata = {
"user_api_key": "sk-test",
"user_api_key_team_id": "team-123",
"user_api_key_user_id": "user-456",
"litellm_call_id": "call-789",
"should_not_propagate": "secret",
}
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
new_callable=AsyncMock,
return_value=mock_response,
) as mock_call,
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
metadata=parent_metadata,
)
call_kwargs = mock_call.call_args.kwargs
propagated = call_kwargs["metadata"]
assert propagated["user_api_key"] == "sk-test"
assert propagated["user_api_key_team_id"] == "team-123"
assert "should_not_propagate" not in propagated
# ---------------------------------------------------------------------------
# Editor: error paths
# ---------------------------------------------------------------------------
async def test_summary_call_failed():
"""When the summary model raises, applied_edits[0].error == 'summary_call_failed'."""
messages = _simple_messages()
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
new_callable=AsyncMock,
side_effect=RuntimeError("network error"),
),
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
assert result.applied_edits[0]["error"] == "summary_call_failed"
assert result.compaction_block is None
assert result.iterations_usage is None
# Messages passed through (at minimum sliced, no compaction blocks)
for msg in result.messages:
content = msg.get("content")
if isinstance(content, list):
for block in content:
assert block.get("type") != "compaction"
async def test_summary_extraction_failed_no_tags():
"""When summary response has no <summary> tags, applied_edits[0].error == 'summary_extraction_failed'."""
messages = _simple_messages()
mock_response = _make_mock_response("I cannot summarize that.")
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
new_callable=AsyncMock,
return_value=mock_response,
),
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
assert result.applied_edits[0]["error"] == "summary_extraction_failed"
assert result.compaction_block is None
assert result.iterations_usage is None
# ---------------------------------------------------------------------------
# Editor: warnings
# ---------------------------------------------------------------------------
async def test_pause_after_compaction_ignored_warning():
"""pause_after_compaction: true → warning recorded, request proceeds normally."""
messages = _simple_messages()
with patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value=None,
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec={
"type": "compact_20260112",
"pause_after_compaction": True,
},
)
edit = result.applied_edits[0]
assert "pause_after_compaction_ignored" in (edit.get("warnings") or [])
# Request still proceeds (here it hits opt-in gate because no model configured)
assert edit.get("error") == "summary_model_not_configured"
async def test_unsupported_trigger_type_falls_back_to_default():
messages = _simple_messages()
with patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value=None,
):
result = await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec={
"type": "compact_20260112",
"trigger": {"type": "output_tokens", "value": 200_000},
},
)
edit = result.applied_edits[0]
warnings = edit.get("warnings") or []
assert any("unsupported_trigger_type" in w for w in warnings)
# ---------------------------------------------------------------------------
# Editor: custom instructions
# ---------------------------------------------------------------------------
async def test_custom_instructions_used_verbatim():
"""Custom instructions are used as-is; the default prompt is NOT appended."""
messages = _simple_messages()
tools = [{"name": "search", "description": "Search tool"}]
mock_response = _make_mock_response("<summary>Custom summary</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=tools,
system=None,
edit_spec={
"type": "compact_20260112",
"instructions": "Summarize everything briefly.",
},
)
assert len(captured_calls) == 1
summary_messages = captured_calls[0]["summary_messages"]
# The last message should be the custom instruction prompt
last_msg = summary_messages[-1]
assert last_msg["role"] == "user"
assert last_msg["content"] == "Summarize everything briefly."
# The "do not call tools" suffix should NOT be in the prompt since custom was set
assert "tool" not in last_msg["content"].lower()
async def test_default_instructions_appended_with_no_tool_suffix_when_no_tools():
"""Without tools, default prompt is used but the no-tool-calls suffix is absent."""
messages = _simple_messages()
mock_response = _make_mock_response("<summary>Default summary</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=None,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
prompt = captured_calls[0]["summary_messages"][-1]["content"]
# Should not contain the no-tool-calls guidance
assert "do not call" not in prompt.lower()
async def test_default_instructions_with_tools_appends_no_tool_suffix():
"""With tools and no custom instructions, the no-tool-calls suffix is appended."""
messages = _simple_messages()
tools = [{"name": "search"}]
mock_response = _make_mock_response("<summary>Tool-aware summary</summary>")
captured_calls: list = []
async def _fake_call_summary_model(**kwargs):
captured_calls.append(kwargs)
return mock_response
with (
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value="claude-haiku-4-5",
),
patch("litellm.token_counter", return_value=200_000),
patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model",
side_effect=_fake_call_summary_model,
),
):
await apply_compact_20260112(
model=MODEL,
messages=messages,
tools=tools,
system=None,
edit_spec=_EDIT_SPEC_DEFAULT,
)
prompt = captured_calls[0]["summary_messages"][-1]["content"]
assert "tool" in prompt.lower()
# ---------------------------------------------------------------------------
# Dispatcher integration: compact_20260112 via apply_context_management
# ---------------------------------------------------------------------------
async def test_dispatcher_routes_compact_edit():
"""compact_20260112 in the dispatcher resolves to opt-in gate when no model set."""
messages = _simple_messages()
with patch(
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
return_value=None,
):
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
system=None,
context_management_spec={"edits": [{"type": "compact_20260112"}]},
)
assert len(result.applied_edits) == 1
assert result.applied_edits[0]["type"] == "compact_20260112"
assert result.applied_edits[0].get("error") == "summary_model_not_configured"
async def test_dispatcher_trigger_below_minimum_raises_through():
"""AnthropicContextManagementError from the editor bubbles up through the dispatcher."""
with pytest.raises(AnthropicContextManagementError):
await apply_context_management(
model=MODEL,
messages=_simple_messages(),
tools=None,
system=None,
context_management_spec={
"edits": [
{
"type": "compact_20260112",
"trigger": {"type": "input_tokens", "value": 1_000},
}
]
},
)
# ---------------------------------------------------------------------------
# _run_polyfill_if_enabled: drop_params gate
# ---------------------------------------------------------------------------
async def test_run_polyfill_skipped_when_drop_params_true():
"""When drop_params=True the polyfill must be skipped (returns None)."""
from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
_run_polyfill_if_enabled,
)
result = await _run_polyfill_if_enabled(
model=MODEL,
messages=_simple_messages(),
tools=None,
system=None,
context_management_spec={"edits": [{"type": "compact_20260112"}]},
metadata={},
drop_params=True,
llm_router=None,
)
assert result is None
async def test_run_polyfill_skipped_when_spec_empty():
"""Empty context_management_spec must also return None (no polyfill work)."""
from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
_run_polyfill_if_enabled,
)
result = await _run_polyfill_if_enabled(
model=MODEL,
messages=_simple_messages(),
tools=None,
system=None,
context_management_spec=None,
metadata={},
drop_params=False,
llm_router=None,
)
assert result is None
# ---------------------------------------------------------------------------
# Endpoint error format: AnthropicContextManagementError → Anthropic 400 body
# ---------------------------------------------------------------------------
def test_anthropic_context_management_error_format():
"""AnthropicContextManagementError must produce an Anthropic-format body via
AnthropicExceptionMapping.transform_to_anthropic_error — the same path the
/v1/messages endpoint takes when it catches this exception."""
from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping
body = AnthropicExceptionMapping.transform_to_anthropic_error(
status_code=400,
raw_message="trigger.value must be at least 50000 tokens",
request_id=None,
)
assert body["type"] == "error"
assert body["error"]["type"] == "invalid_request_error"
assert "50000" in body["error"]["message"]
def test_anthropic_context_management_error_attrs():
"""AnthropicContextManagementError carries status_code and message correctly."""
err = AnthropicContextManagementError(
status_code=400,
message="trigger.value must be at least 50000 tokens",
)
assert err.status_code == 400
assert "50000" in err.message
# ---------------------------------------------------------------------------
# Endpoint integration: /v1/messages → Anthropic 400 on context management error
# ---------------------------------------------------------------------------
def test_endpoint_returns_anthropic_400_on_context_management_error():
"""The /v1/messages endpoint must catch AnthropicContextManagementError and
return an Anthropic-format 400 JSONResponse — not a 500 ProxyException."""
import sys
from unittest.mock import AsyncMock, MagicMock, patch
from fastapi import FastAPI
from fastapi.testclient import TestClient
from litellm.proxy.anthropic_endpoints.endpoints import router
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
# Stub proxy_server to avoid apscheduler/heavy proxy deps imported lazily
# inside the route handler at request time.
mock_proxy_server = MagicMock()
mock_proxy_server.general_settings = {}
mock_proxy_server.llm_router = None
mock_proxy_server.proxy_config = MagicMock()
mock_proxy_server.proxy_logging_obj = MagicMock()
mock_proxy_server.user_api_base = None
mock_proxy_server.user_max_tokens = None
mock_proxy_server.user_model = None
mock_proxy_server.user_request_timeout = None
mock_proxy_server.user_temperature = None
mock_proxy_server.version = "test"
with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}):
with patch(
"litellm.proxy.anthropic_endpoints.endpoints.ProxyBaseLLMRequestProcessing"
) as mock_cls:
mock_instance = MagicMock()
mock_instance.base_process_llm_request = AsyncMock(
side_effect=AnthropicContextManagementError(
status_code=400,
message="trigger.value must be at least 50000 tokens",
)
)
mock_cls.return_value = mock_instance
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = lambda: MagicMock()
client = TestClient(app, raise_server_exceptions=False)
response = client.post(
"/v1/messages",
json={
"model": "gpt-4o",
"messages": [{"role": "user", "content": "hi"}],
},
headers={"Authorization": "Bearer test-key"},
)
assert response.status_code == 400
body = response.json()
assert body["type"] == "error"
assert body["error"]["type"] == "invalid_request_error"
assert "50000" in body["error"]["message"]

View file

@ -43,9 +43,9 @@ def _history_with_two_tool_pairs():
]
def test_unknown_edit_type_is_noop():
async def test_unknown_edit_type_is_noop():
messages = _history_with_two_tool_pairs()
new_messages, applied = apply_context_management(
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@ -54,13 +54,13 @@ def test_unknown_edit_type_is_noop():
"edits": [{"type": "totally_not_a_real_edit_20999999"}]
},
)
assert applied == []
assert new_messages == messages
assert result.applied_edits == []
assert result.messages == messages
def test_known_edit_is_applied():
async def test_known_edit_is_applied():
messages = _history_with_two_tool_pairs()
_, applied = apply_context_management(
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@ -75,14 +75,14 @@ def test_known_edit_is_applied():
]
},
)
assert len(applied) == 1
assert applied[0]["type"] == "clear_tool_uses_20250919"
assert applied[0]["cleared_tool_uses"] == 1
assert len(result.applied_edits) == 1
assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919"
assert result.applied_edits[0]["cleared_tool_uses"] == 1
def test_mixed_known_unknown_only_known_applied():
async def test_mixed_known_unknown_only_known_applied():
messages = _history_with_two_tool_pairs()
_, applied = apply_context_management(
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
@ -99,33 +99,33 @@ def test_mixed_known_unknown_only_known_applied():
]
},
)
assert len(applied) == 1
assert applied[0]["type"] == "clear_tool_uses_20250919"
assert len(result.applied_edits) == 1
assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919"
def test_empty_or_missing_edits_list():
async def test_empty_or_missing_edits_list():
messages = _history_with_two_tool_pairs()
for spec in [{}, {"edits": None}, {"edits": []}, None]:
new_messages, applied = apply_context_management(
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
system=None,
context_management_spec=spec, # type: ignore[arg-type]
)
assert applied == []
assert new_messages == messages
assert result.applied_edits == []
assert result.messages == messages
def test_malformed_edit_entries_are_skipped():
async def test_malformed_edit_entries_are_skipped():
"""Non-dict entries in `edits` list should be silently skipped."""
messages = _history_with_two_tool_pairs()
new_messages, applied = apply_context_management(
result = await apply_context_management(
model=MODEL,
messages=messages,
tools=None,
system=None,
context_management_spec={"edits": ["not a dict", 42, None, {"type": None}]},
)
assert applied == []
assert new_messages == messages
assert result.applied_edits == []
assert result.messages == messages