diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 6908921a43f..7afe3334f29 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -12,13 +12,18 @@ from typing import ( ) import litellm +from litellm._logging import verbose_logger from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( AnthropicAdapter, ) +from litellm.llms.anthropic.experimental_pass_through.context_management import ( + AnthropicContextManagementError, + PolyfillResult, + apply_context_management, +) from litellm.llms.anthropic.experimental_pass_through.utils import ( is_reasoning_auto_summary_enabled, ) -from litellm.types.llms.anthropic import AppliedEdit from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, ) @@ -31,9 +36,58 @@ if TYPE_CHECKING: # Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge. ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset( - {"output_config", "context_management", "_polyfill_applied_edits"} + {"output_config", "context_management", "_polyfill_result"} ) + +async def _run_polyfill_if_enabled( + *, + model: str, + messages: List[Dict], + tools: Optional[List[Dict]], + system: Optional[Any], + context_management_spec: Any, + metadata: Optional[Dict], + drop_params: Optional[bool], + llm_router: Any, +) -> Optional[PolyfillResult]: + """Run the async context_management polyfill if a spec is present. + + Returns ``None`` when the spec is empty or drop_params is on. Raises + ``AnthropicContextManagementError`` so the /v1/messages endpoint can + emit an Anthropic-format 400. All other exceptions are best-effort + swallowed (matches v0 behavior). + """ + if not context_management_spec: + return None + + effective_drop_params = ( + drop_params if drop_params is not None else litellm.drop_params + ) + if effective_drop_params: + return None + + try: + return await apply_context_management( + model=model, + messages=messages, + tools=tools, + system=system, + context_management_spec=context_management_spec, + metadata=metadata, + llm_router=llm_router, + ) + except AnthropicContextManagementError: + # Surface validation errors so the endpoint can emit an Anthropic-format + # 400. Other exception types fall into the best-effort branch below. + raise + except Exception as e: + verbose_logger.exception( + "context_management polyfill: skipping edits due to error: %s", e + ) + return None + + ######################################################## # init adapter ANTHROPIC_ADAPTER = AnthropicAdapter() @@ -303,21 +357,49 @@ class LiteLLMMessagesToCompletionTransformationHandler: top_k: Optional[int] = None, top_p: Optional[float] = None, output_format: Optional[Dict] = None, - _polyfill_applied_edits: Optional[List[AppliedEdit]] = None, **kwargs, ) -> Union[AnthropicMessagesResponse, AsyncIterator]: """Handle non-Anthropic models asynchronously using the adapter""" + context_management = kwargs.pop("context_management", None) + drop_params: Optional[bool] = kwargs.get("drop_params", None) + litellm_router = kwargs.pop("litellm_router", None) + if litellm_router is None: + try: + from litellm.proxy.proxy_server import llm_router as _proxy_router + + litellm_router = _proxy_router + except Exception: + pass + + polyfill_result = await _run_polyfill_if_enabled( + model=model, + messages=messages, + tools=tools, + system=system, + context_management_spec=context_management, + metadata=metadata, + drop_params=drop_params, + llm_router=litellm_router, + ) + + effective_messages = ( + polyfill_result.messages if polyfill_result is not None else messages + ) + effective_system = ( + polyfill_result.system if polyfill_result is not None else system + ) + ( completion_kwargs, tool_name_mapping, ) = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( max_tokens=max_tokens, - messages=messages, + messages=effective_messages, model=model, metadata=metadata, stop_sequences=stop_sequences, stream=stream, - system=system, + system=effective_system, temperature=temperature, thinking=thinking, tool_choice=tool_choice, @@ -336,7 +418,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: completion_response, model=model, tool_name_mapping=tool_name_mapping, - applied_edits=_polyfill_applied_edits, + polyfill_result=polyfill_result, ) ) if transformed_stream is not None: @@ -346,7 +428,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params( cast(ModelResponse, completion_response), tool_name_mapping=tool_name_mapping, - applied_edits=_polyfill_applied_edits, + polyfill_result=polyfill_result, ) if anthropic_response is not None: return anthropic_response @@ -369,7 +451,6 @@ class LiteLLMMessagesToCompletionTransformationHandler: top_p: Optional[float] = None, output_format: Optional[Dict] = None, _is_async: bool = False, - _polyfill_applied_edits: Optional[List[AppliedEdit]] = None, **kwargs, ) -> Union[ AnthropicMessagesResponse, @@ -393,7 +474,6 @@ class LiteLLMMessagesToCompletionTransformationHandler: top_k=top_k, top_p=top_p, output_format=output_format, - _polyfill_applied_edits=_polyfill_applied_edits, **kwargs, ) @@ -426,7 +506,6 @@ class LiteLLMMessagesToCompletionTransformationHandler: completion_response, model=model, tool_name_mapping=tool_name_mapping, - applied_edits=_polyfill_applied_edits, ) ) if transformed_stream is not None: @@ -436,7 +515,6 @@ class LiteLLMMessagesToCompletionTransformationHandler: anthropic_response = ANTHROPIC_ADAPTER.translate_completion_output_params( cast(ModelResponse, completion_response), tool_name_mapping=tool_name_mapping, - applied_edits=_polyfill_applied_edits, ) if anthropic_response is not None: return anthropic_response diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 9fc88874b13..4069bd7fd31 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -75,6 +75,9 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( from litellm.litellm_core_utils.prompt_templates.factory import ( THOUGHT_SIGNATURE_SEPARATOR, ) +from litellm.llms.anthropic.experimental_pass_through.context_management import ( + PolyfillResult, +) from litellm.types.llms.anthropic import ( ANTHROPIC_HOSTED_TOOLS, AllAnthropicToolsValues, @@ -88,6 +91,7 @@ from litellm.types.llms.anthropic import ( AnthropicResponseContentBlockThinking, AnthropicResponseContentBlockToolUse, AppliedEdit, + CompactionBlock, ContentBlockDelta, ContentJsonBlockDelta, ContentTextBlockDelta, @@ -97,6 +101,7 @@ from litellm.types.llms.anthropic import ( MessageBlockDelta, MessageDelta, UsageDelta, + UsageIteration, ) from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, @@ -197,7 +202,7 @@ class AnthropicAdapter: self, response: ModelResponse, tool_name_mapping: Optional[Dict[str, str]] = None, - applied_edits: Optional[List[AppliedEdit]] = None, + polyfill_result: Optional[PolyfillResult] = None, ) -> Optional[AnthropicMessagesResponse]: """ Translate OpenAI response to Anthropic format. @@ -207,12 +212,12 @@ class AnthropicAdapter: tool_name_mapping: Optional mapping of truncated tool names to original names. Used to restore original names for tools that exceeded OpenAI's 64-char limit. - applied_edits: Polyfill AppliedEdit list for response context_management. + polyfill_result: PolyfillResult from context_management polyfill. """ return LiteLLMAnthropicMessagesAdapter().translate_openai_response_to_anthropic( response=response, tool_name_mapping=tool_name_mapping, - applied_edits=applied_edits, + polyfill_result=polyfill_result, ) def translate_completion_output_params_streaming( @@ -220,7 +225,7 @@ class AnthropicAdapter: completion_stream: Any, model: str, tool_name_mapping: Optional[Dict[str, str]] = None, - applied_edits: Optional[List[AppliedEdit]] = None, + polyfill_result: Optional[PolyfillResult] = None, ) -> Union[AsyncIterator[bytes], None]: """ Translate OpenAI streaming response to Anthropic format. @@ -229,8 +234,9 @@ class AnthropicAdapter: completion_stream: The OpenAI streaming response model: The model name tool_name_mapping: Optional mapping of truncated tool names to original names. - applied_edits: Polyfill AppliedEdit list on final message_delta. + polyfill_result: PolyfillResult from context_management polyfill. """ + applied_edits = polyfill_result.applied_edits if polyfill_result else None anthropic_wrapper = AnthropicStreamWrapper( completion_stream=completion_stream, model=model, @@ -1350,7 +1356,7 @@ class LiteLLMAnthropicMessagesAdapter: self, response: ModelResponse, tool_name_mapping: Optional[Dict[str, str]] = None, - applied_edits: Optional[List[AppliedEdit]] = None, + polyfill_result: Optional[PolyfillResult] = None, ) -> AnthropicMessagesResponse: """ Translate OpenAI response to Anthropic format. @@ -1360,13 +1366,17 @@ class LiteLLMAnthropicMessagesAdapter: tool_name_mapping: Optional mapping of truncated tool names to original names. Used to restore original names for tools that exceeded OpenAI's 64-char limit. - applied_edits: Polyfill AppliedEdit list for response context_management. + polyfill_result: PolyfillResult from context_management polyfill. """ ## translate content block anthropic_content = self._translate_openai_content_to_anthropic( choices=response.choices, # type: ignore tool_name_mapping=tool_name_mapping, ) + + if polyfill_result is not None and polyfill_result.compaction_block is not None: + anthropic_content.insert(0, polyfill_result.compaction_block) # type: ignore[arg-type] + ## extract finish reason anthropic_finish_reason = self._translate_openai_finish_reason_to_anthropic( openai_finish_reason=response.choices[0].finish_reason # type: ignore @@ -1395,6 +1405,14 @@ class LiteLLMAnthropicMessagesAdapter: if cached_tokens > 0: anthropic_usage["cache_read_input_tokens"] = cached_tokens + if polyfill_result is not None and polyfill_result.iterations_usage is not None: + message_iteration: UsageIteration = { + "type": "message", + "input_tokens": uncached_input_tokens, + "output_tokens": usage.completion_tokens or 0, + } + anthropic_usage["iterations"] = list(polyfill_result.iterations_usage) + [message_iteration] # type: ignore[typeddict-unknown-key] + translated_obj = AnthropicMessagesResponse( id=response.id, type="message", @@ -1406,6 +1424,7 @@ class LiteLLMAnthropicMessagesAdapter: stop_reason=anthropic_finish_reason, ) + applied_edits = polyfill_result.applied_edits if polyfill_result else None if applied_edits: translated_obj["context_management"] = ContextManagementResponse( applied_edits=list(applied_edits) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py b/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py index f9bd9f4c66d..729b2864524 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/__init__.py @@ -1,4 +1,11 @@ from .constants import CLEARED_TOOL_RESULT_PLACEHOLDER from .dispatcher import apply_context_management +from .errors import AnthropicContextManagementError +from .result import PolyfillResult -__all__ = ["apply_context_management", "CLEARED_TOOL_RESULT_PLACEHOLDER"] +__all__ = [ + "apply_context_management", + "AnthropicContextManagementError", + "CLEARED_TOOL_RESULT_PLACEHOLDER", + "PolyfillResult", +] diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py index 4003591c0d4..0d165698f23 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py @@ -6,3 +6,28 @@ DEFAULT_INPUT_TOKENS_TRIGGER = 100_000 DEFAULT_KEEP_TOOL_USES = 3 CLEARED_TOOL_RESULT_PLACEHOLDER = "[Cleared by context management]" + +# compact_20260112 +COMPACT_EDIT_TYPE = "compact_20260112" +COMPACT_DEFAULT_TRIGGER_TOKENS = 150_000 +COMPACT_MIN_TRIGGER_TOKENS = 50_000 +COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model" +COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: " + +# Default summarization prompt from the Anthropic spec. +COMPACT_DEFAULT_INSTRUCTIONS = ( + "You have written a partial transcript for the initial task above. Please " + "write a summary of the transcript. The purpose of this summary is to " + "provide continuity so you can continue to make progress towards solving " + "the task in a future context, where the raw history above may not be " + "accessible and will be replaced with this summary. Write down anything " + "that would be helpful, including the state, next steps, learnings etc. " + "You must wrap your summary in a block." +) + +# Appended to the default prompt when ``tools`` are present and the caller +# did not supply custom ``instructions``. Matches the guidance in the +# Anthropic docs under "Compaction might fail when tools are defined". +COMPACT_NO_TOOL_CALLS_SUFFIX = ( + " Do not call any tools while writing this summary; respond with text only." +) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py b/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py index 93881580e92..ab3a905c64c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/dispatcher.py @@ -1,56 +1,85 @@ """Dispatch ``context_management`` edits to registered polyfill editors.""" -from typing import Any, Callable, Dict, List, Optional, Tuple, Union +import inspect +from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple, Union, cast from litellm._logging import verbose_logger -from litellm.types.llms.anthropic import AppliedEdit -from .constants import CLEAR_TOOL_USES_EDIT_TYPE -from .editors import apply_clear_tool_uses_20250919 +from .constants import CLEAR_TOOL_USES_EDIT_TYPE, COMPACT_EDIT_TYPE +from .editors import apply_clear_tool_uses_20250919, apply_compact_20260112 +from .result import PolyfillResult -EditorFn = Callable[..., Tuple[List[Dict[str, Any]], Optional[AppliedEdit]]] +EditorFn = Callable[..., Any] _EDITOR_REGISTRY: Dict[str, EditorFn] = { CLEAR_TOOL_USES_EDIT_TYPE: apply_clear_tool_uses_20250919, + COMPACT_EDIT_TYPE: apply_compact_20260112, } -def apply_context_management( +def _normalize_spec( + spec: Union[Dict[str, Any], List[Dict[str, Any]], None], +) -> Optional[List[Dict[str, Any]]]: + """Accept Anthropic-native dict form or OpenAI list form; return edits list.""" + if isinstance(spec, list): + # Local import to avoid an import cycle at module load. + from litellm.llms.anthropic.chat.transformation import AnthropicConfig + + spec = AnthropicConfig.map_openai_context_management_to_anthropic(spec) + + edits = spec.get("edits") if isinstance(spec, dict) else None + if not edits or not isinstance(edits, list): + return None + return [edit for edit in edits if isinstance(edit, dict)] + + +def _wrap_editor_return(raw: Any, *, fallback_system: Any) -> PolyfillResult: + """Coerce an editor's native return shape into a ``PolyfillResult``. + + v0 sync editors (e.g. ``clear_tool_uses_20250919``) return a 2-tuple + ``(messages, Optional[AppliedEdit])``. The new async ``compact_20260112`` + editor returns a ``PolyfillResult`` directly. + """ + if isinstance(raw, PolyfillResult): + return raw + # Legacy 2-tuple return — sync editors don't mutate ``system``, so + # carry the caller's value forward. + messages, applied = cast(Tuple[List[Dict[str, Any]], Any], raw) + return PolyfillResult( + messages=messages, + system=fallback_system, + applied_edits=[applied] if applied is not None else [], + ) + + +async def apply_context_management( *, model: str, messages: List[Dict[str, Any]], tools: Optional[List[Dict[str, Any]]], system: Any, context_management_spec: Union[Dict[str, Any], List[Dict[str, Any]], None], -) -> Tuple[List[Dict[str, Any]], List[AppliedEdit]]: - """Run edits in order; return (messages, applied_edits that fired).""" - # Accept both Anthropic-native dict form and OpenAI list form. The other - # provider paths normalize via ``map_openai_context_management_to_anthropic`` - # before dispatching; do the same here so the polyfill path doesn't silently - # no-op on list input. - if isinstance(context_management_spec, list): - from litellm.llms.anthropic.chat.transformation import AnthropicConfig + metadata: Optional[Dict[str, Any]] = None, + llm_router: Any = None, +) -> PolyfillResult: + """Run edits in order; return a single ``PolyfillResult``. - context_management_spec = ( - AnthropicConfig.map_openai_context_management_to_anthropic( - context_management_spec - ) - ) + The dispatcher is async so async editors (``compact_20260112``) can + ``await`` the configured summarization model. Sync editors are called + inline — ``inspect.iscoroutinefunction`` decides how each editor is + invoked. + """ + edits = _normalize_spec(context_management_spec) + if not edits: + return PolyfillResult(messages=messages, system=system, applied_edits=[]) - edits = ( - context_management_spec.get("edits") - if isinstance(context_management_spec, dict) - else None - ) - if not edits or not isinstance(edits, list): - return messages, [] - - applied_edits: List[AppliedEdit] = [] current_messages = messages + current_system = system + aggregated_applied: List[Dict[str, Any]] = [] + aggregated_compaction_block = None + aggregated_iterations_usage = None for edit_spec in edits: - if not isinstance(edit_spec, dict): - continue edit_type = edit_spec.get("type") editor = _EDITOR_REGISTRY.get(edit_type) if isinstance(edit_type, str) else None if editor is None: @@ -60,14 +89,36 @@ def apply_context_management( ) continue - current_messages, applied = editor( - model=model, - messages=current_messages, - tools=tools, - system=system, - edit_spec=edit_spec, - ) - if applied is not None: - applied_edits.append(applied) + kwargs: Dict[str, Any] = { + "model": model, + "messages": current_messages, + "tools": tools, + "system": current_system, + "edit_spec": edit_spec, + } + # Only async editors accept these — passing them to sync v0 editors + # would break their signature. + if inspect.iscoroutinefunction(editor): + kwargs["metadata"] = metadata + kwargs["llm_router"] = llm_router + raw_result = await cast(Callable[..., Awaitable[Any]], editor)(**kwargs) + else: + raw_result = editor(**kwargs) - return current_messages, applied_edits + result = _wrap_editor_return(raw_result, fallback_system=current_system) + + current_messages = result.messages + current_system = result.system + aggregated_applied.extend(result.applied_edits) + if result.compaction_block is not None: + aggregated_compaction_block = result.compaction_block + if result.iterations_usage is not None: + aggregated_iterations_usage = result.iterations_usage + + return PolyfillResult( + messages=current_messages, + system=current_system, + applied_edits=cast(List[Any], aggregated_applied), + compaction_block=aggregated_compaction_block, + iterations_usage=aggregated_iterations_usage, + ) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py index d402632380d..3e933a9880a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/__init__.py @@ -1,3 +1,4 @@ from .clear_tool_uses import apply_clear_tool_uses_20250919 +from .compact import apply_compact_20260112 -__all__ = ["apply_clear_tool_uses_20250919"] +__all__ = ["apply_clear_tool_uses_20250919", "apply_compact_20260112"] diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py new file mode 100644 index 00000000000..62e4e2c3602 --- /dev/null +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -0,0 +1,494 @@ +"""``compact_20260112`` polyfill (server-side context compaction). + +Mirrors Anthropic's native ``compact_20260112`` for non-Anthropic providers: + +- Scans the message history for an existing ``compaction`` block; everything + before it is dropped (slice). +- If still over the configured trigger, calls a separately-configured + summarization model and synthesizes a fresh ``compaction`` block. +- The summary is injected as a system-message prefix on the downstream call + (the user/assistant log carries no ``compaction`` block downstream). +- The synthesized ``compaction`` block is returned via ``PolyfillResult`` so + the response adapter can prepend it to the response ``content`` array. +""" + +import re +from typing import Any, Dict, List, Optional, Tuple, Union, cast + +import litellm +from litellm._logging import verbose_logger +from litellm.types.llms.anthropic import ( + AppliedEdit, + CompactionBlock, + UsageIteration, +) + +from ..constants import ( + COMPACT_DEFAULT_INSTRUCTIONS, + COMPACT_DEFAULT_TRIGGER_TOKENS, + COMPACT_EDIT_TYPE, + COMPACT_MIN_TRIGGER_TOKENS, + COMPACT_NO_TOOL_CALLS_SUFFIX, + COMPACT_SUMMARY_MODEL_SETTING_KEY, + COMPACT_SUMMARY_SYSTEM_PREFIX, +) +from ..errors import AnthropicContextManagementError +from ..result import PolyfillResult + +# Auth metadata fields propagated from the parent request to the summary call +# so the summary's spend is attributed to the same team/key. The list mirrors +# the fields populated by +# ``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata``. +_PROPAGATED_METADATA_KEYS = ( + "user_api_key", + "user_api_key_alias", + "user_api_key_team_id", + "user_api_key_team_alias", + "user_api_key_user_id", + "user_api_key_user_email", + "user_api_key_org_id", + "litellm_call_id", + "litellm_parent_otel_span", +) + +_SUMMARY_TAG_RE = re.compile(r"(.*?)", re.IGNORECASE | re.DOTALL) + + +def _read_summary_model_setting() -> Optional[str]: + """Look up the configured summarization model from proxy general_settings.""" + try: + from litellm.proxy.proxy_server import general_settings + except Exception: + return None + value = general_settings.get(COMPACT_SUMMARY_MODEL_SETTING_KEY) + return value if isinstance(value, str) and value else None + + +def _find_latest_compaction_index( + messages: List[Dict[str, Any]], +) -> Tuple[Optional[int], Optional[int]]: + """Return (message_index, block_index) of the most recent compaction block. + + ``None, None`` if no compaction block is present. Iterates from the end so + only the latest one is considered. + """ + for msg_idx in range(len(messages) - 1, -1, -1): + content = messages[msg_idx].get("content") + if not isinstance(content, list): + continue + for blk_idx in range(len(content) - 1, -1, -1): + block = content[blk_idx] + if isinstance(block, dict) and block.get("type") == "compaction": + return msg_idx, blk_idx + return None, None + + +def _slice_around_compaction_block( + messages: List[Dict[str, Any]], +) -> Tuple[List[Dict[str, Any]], Optional[Dict[str, Any]]]: + """Apply Anthropic's "drop everything before the compaction block" rule. + + Returns ``(sliced_messages_with_compaction_block, compaction_block_dict)`` + if a block was found, else ``(original_messages, None)``. The sliced result + keeps the compaction block in the assistant turn that originally carried + it (in practice it's the only block in that turn) so callers can still + extract the summary text from it. + """ + msg_idx, blk_idx = _find_latest_compaction_index(messages) + if msg_idx is None or blk_idx is None: + return messages, None + + original_msg = messages[msg_idx] + original_content = original_msg["content"] + compaction_block = cast(Dict[str, Any], original_content[blk_idx]) + + # Per Anthropic's contract everything before the compaction block is + # dropped, including earlier blocks within the same assistant message. + sliced_content = list(original_content[blk_idx:]) + sliced_first_msg = {**original_msg, "content": sliced_content} + + sliced_messages: List[Dict[str, Any]] = [sliced_first_msg] + sliced_messages.extend(messages[msg_idx + 1 :]) + return sliced_messages, compaction_block + + +def _strip_compaction_blocks( + messages: List[Dict[str, Any]], +) -> List[Dict[str, Any]]: + """Drop any ``compaction`` content blocks from messages. + + Used to build the downstream-bound message list — the adapter has no + concept of a compaction block, so it must not see one. + """ + cleaned: List[Dict[str, Any]] = [] + for msg in messages: + content = msg.get("content") + if not isinstance(content, list): + cleaned.append(msg) + continue + filtered = [ + block + for block in content + if not (isinstance(block, dict) and block.get("type") == "compaction") + ] + if not filtered: + # The compaction block was the only content; drop the whole turn. + continue + cleaned.append({**msg, "content": filtered}) + return cleaned + + +def _augment_system_with_summary( + system: Optional[Union[str, List[Dict[str, Any]]]], + summary_text: str, +) -> Union[str, List[Dict[str, Any]]]: + """Prepend a "Previous conversation summary: ..." block to ``system``.""" + prefix = f"{COMPACT_SUMMARY_SYSTEM_PREFIX}{summary_text}\n\n" + if system is None: + return prefix.rstrip() + if isinstance(system, str): + return f"{prefix}{system}" + # List of content blocks: prepend the prefix to the first text block, + # otherwise insert a new text block at the head. + for idx, block in enumerate(system): + if isinstance(block, dict) and block.get("type") == "text": + existing = block.get("text", "") or "" + new_block = {**block, "text": f"{prefix}{existing}"} + return [*system[:idx], new_block, *system[idx + 1 :]] + return [{"type": "text", "text": prefix.rstrip()}, *system] + + +def _resolve_trigger_tokens(edit_spec: Dict[str, Any]) -> Tuple[int, List[str]]: + """Validate and resolve ``trigger.value``. + + Raises ``AnthropicContextManagementError`` if the explicitly-supplied value + is below the 50k minimum. Unknown ``trigger.type`` values fall back to + ``input_tokens`` with a warning. + """ + warnings: List[str] = [] + trigger = edit_spec.get("trigger") or {} + if not isinstance(trigger, dict): + warnings.append("trigger_not_a_dict_using_default") + return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings + + trigger_type = trigger.get("type", "input_tokens") + if trigger_type != "input_tokens": + warnings.append(f"unsupported_trigger_type_{trigger_type}_using_input_tokens") + + value = trigger.get("value") + if value is None: + return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings + if not isinstance(value, int): + warnings.append("trigger_value_not_int_using_default") + return COMPACT_DEFAULT_TRIGGER_TOKENS, warnings + if value < COMPACT_MIN_TRIGGER_TOKENS: + raise AnthropicContextManagementError( + status_code=400, + message=( + f"context_management.compact_20260112.trigger.value must be at " + f"least {COMPACT_MIN_TRIGGER_TOKENS} tokens" + ), + ) + return value, warnings + + +def _build_summary_prompt( + edit_spec: Dict[str, Any], tools: Optional[List[Dict[str, Any]]] +) -> str: + custom = edit_spec.get("instructions") + if isinstance(custom, str) and custom.strip(): + return custom + prompt = COMPACT_DEFAULT_INSTRUCTIONS + if tools: + prompt = f"{prompt}{COMPACT_NO_TOOL_CALLS_SUFFIX}" + return prompt + + +def _propagate_metadata(parent_metadata: Optional[Dict[str, Any]]) -> Dict[str, Any]: + if not parent_metadata: + return {} + propagated: Dict[str, Any] = {} + for key in _PROPAGATED_METADATA_KEYS: + if key in parent_metadata: + propagated[key] = parent_metadata[key] + return propagated + + +def _count_effective_tokens( + model: str, + effective_messages: List[Dict[str, Any]], + compaction_block: Optional[Dict[str, Any]], + tools: Optional[List[Dict[str, Any]]], +) -> int: + """Token-count the conversation as it will appear downstream. + + The compaction block (if any) becomes a system prefix on the downstream + call, so its content still counts even though it isn't in ``messages``. + """ + # Local import to avoid pulling the adapter at module load time. + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + messages_without_compaction = _strip_compaction_blocks(effective_messages) + try: + openai_shape = ( + LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai( + messages=cast(Any, messages_without_compaction) + ) + ) + except Exception as e: + verbose_logger.debug( + "compact_20260112: anthropic→openai translation failed during token " + "count, falling back to raw messages: %s", + e, + ) + openai_shape = cast(Any, messages_without_compaction) + + total = litellm.token_counter( + model=model, + messages=cast(Any, openai_shape), + tools=cast(Any, tools), + ) + if compaction_block is not None: + content = compaction_block.get("content") or "" + if content: + total += litellm.token_counter(model=model, text=content) + return total + + +def _extract_summary_text(raw: Optional[str]) -> Optional[str]: + if not raw: + return None + match = _SUMMARY_TAG_RE.search(raw) + if match is None: + return None + summary = match.group(1).strip() + return summary or None + + +def _build_summary_messages( + effective_messages: List[Dict[str, Any]], + prompt: str, +) -> List[Dict[str, Any]]: + """Build the OpenAI-shape message list for the summary call. + + The conversation history is translated to OpenAI shape; the + summarization prompt is appended as a final user turn. + """ + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + stripped = _strip_compaction_blocks(effective_messages) + try: + openai_messages = ( + LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai( + messages=cast(Any, stripped) + ) + ) + except Exception as e: + verbose_logger.warning( + "compact_20260112: anthropic→openai translation failed when " + "building summary call; falling back to raw shape: %s", + e, + ) + openai_messages = cast(Any, stripped) + + return [*openai_messages, {"role": "user", "content": prompt}] + + +async def _call_summary_model( + *, + summary_model: str, + summary_messages: List[Dict[str, Any]], + metadata: Dict[str, Any], + llm_router: Any, +) -> Any: + """Invoke the configured summary model. + + Prefers ``llm_router.acompletion`` so the model alias resolves against the + proxy's ``model_list``; falls back to ``litellm.acompletion`` if no router + is available (e.g. SDK usage outside the proxy). + """ + call_kwargs: Dict[str, Any] = { + "model": summary_model, + "messages": summary_messages, + "metadata": metadata, + } + if llm_router is not None and hasattr(llm_router, "acompletion"): + return await llm_router.acompletion(**call_kwargs) + return await litellm.acompletion(**call_kwargs) + + +def _extract_response_text(response: Any) -> Optional[str]: + try: + choice = response.choices[0] + message = choice.message + content = getattr(message, "content", None) + if isinstance(content, str): + return content + # Some providers return a list of content parts. + if isinstance(content, list): + text_parts = [ + part.get("text", "") + for part in content + if isinstance(part, dict) and part.get("type") == "text" + ] + return "".join(text_parts) or None + except (AttributeError, IndexError, KeyError): + return None + return None + + +def _extract_usage(response: Any) -> Tuple[int, int]: + usage = getattr(response, "usage", None) + if usage is None: + return 0, 0 + return ( + int(getattr(usage, "prompt_tokens", 0) or 0), + int(getattr(usage, "completion_tokens", 0) or 0), + ) + + +async def apply_compact_20260112( + *, + model: str, + messages: List[Dict[str, Any]], + tools: Optional[List[Dict[str, Any]]], + system: Optional[Union[str, List[Dict[str, Any]]]], + edit_spec: Dict[str, Any], + metadata: Optional[Dict[str, Any]] = None, + llm_router: Any = None, +) -> PolyfillResult: + """Apply ``compact_20260112``; return a ``PolyfillResult``. + + See module docstring for the algorithm. Errors are best-effort: when the + summary call fails or the response is malformed, the editor returns the + pre-summary state (with ``applied_edits[0].error`` populated) so the + original request still proceeds. + """ + # Validation runs first. Raising AnthropicContextManagementError here is + # the only path on which the polyfill aborts the request. + trigger_tokens, warnings = _resolve_trigger_tokens(edit_spec) + if edit_spec.get("pause_after_compaction"): + warnings.append("pause_after_compaction_ignored") + + applied: AppliedEdit = {"type": COMPACT_EDIT_TYPE} + if warnings: + applied["warnings"] = warnings + + # Opt-in gate: no summary model configured → no-op. + summary_model = _read_summary_model_setting() + if summary_model is None: + applied["error"] = "summary_model_not_configured" + return PolyfillResult( + messages=messages, + system=system, + applied_edits=[applied], + ) + + # Phase A: slice around any existing compaction block. + effective_messages, prior_compaction_block = _slice_around_compaction_block( + messages + ) + prior_summary_text = ( + prior_compaction_block.get("content") if prior_compaction_block else None + ) + augmented_system: Union[str, List[Dict[str, Any]], None] = system + if isinstance(prior_summary_text, str) and prior_summary_text: + augmented_system = _augment_system_with_summary(system, prior_summary_text) + + downstream_messages = _strip_compaction_blocks(effective_messages) + + # Phase B: threshold check. + try: + current_tokens = _count_effective_tokens( + model=model, + effective_messages=effective_messages, + compaction_block=prior_compaction_block, + tools=tools, + ) + except Exception as e: + verbose_logger.warning( + "compact_20260112: token_counter failed; assuming under threshold: %s", e + ) + current_tokens = 0 + + verbose_logger.debug( + "compact_20260112: current_tokens=%s trigger=%s", current_tokens, trigger_tokens + ) + + if current_tokens <= trigger_tokens: + # Slice-only path. If Phase A fired we still slice + system-prefix. + return PolyfillResult( + messages=downstream_messages, + system=augmented_system, + applied_edits=[applied], + ) + + # Phase C: summarize. + prompt = _build_summary_prompt(edit_spec, tools) + summary_messages = _build_summary_messages(effective_messages, prompt) + propagated_metadata = _propagate_metadata(metadata) + + try: + response = await _call_summary_model( + summary_model=summary_model, + summary_messages=summary_messages, + metadata=propagated_metadata, + llm_router=llm_router, + ) + except Exception as e: + verbose_logger.warning("compact_20260112: summary call failed: %s", e) + applied["error"] = "summary_call_failed" + return PolyfillResult( + messages=downstream_messages, + system=augmented_system, + applied_edits=[applied], + ) + + summary_text = _extract_summary_text(_extract_response_text(response)) + if summary_text is None: + applied["error"] = "summary_extraction_failed" + return PolyfillResult( + messages=downstream_messages, + system=augmented_system, + applied_edits=[applied], + ) + + summary_input_tokens, summary_output_tokens = _extract_usage(response) + applied["summary_input_tokens"] = summary_input_tokens + applied["summary_output_tokens"] = summary_output_tokens + + compaction_block: CompactionBlock = { + "type": "compaction", + "content": summary_text, + } + iterations_usage: List[UsageIteration] = [ + { + "type": "compaction", + "input_tokens": summary_input_tokens, + "output_tokens": summary_output_tokens, + } + ] + + # Per Anthropic's contract, everything before the compaction block is + # dropped. Phase D: the user/assistant log goes empty; the summary lives + # on the system message instead. Anthropic requires a non-empty messages + # array, so keep the most recent original user turn so the model has the + # question to answer. + summarized_system = _augment_system_with_summary(system, summary_text) + downstream_messages_after_summary: List[Dict[str, Any]] = [] + for msg in reversed(messages): + if msg.get("role") == "user": + downstream_messages_after_summary = [msg] + break + + return PolyfillResult( + messages=downstream_messages_after_summary, + system=summarized_system, + applied_edits=[applied], + compaction_block=compaction_block, + iterations_usage=iterations_usage, + ) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py b/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py new file mode 100644 index 00000000000..1b14089a451 --- /dev/null +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/errors.py @@ -0,0 +1,14 @@ +"""Exceptions raised by the context_management polyfill.""" + + +class AnthropicContextManagementError(Exception): + """Validation error from the polyfill, surfaced as an Anthropic-format 4xx. + + The `/v1/messages` endpoint catches this in its exception handler and + emits an Anthropic-shaped error body instead of the default OpenAI shape. + """ + + def __init__(self, *, status_code: int, message: str) -> None: + super().__init__(message) + self.status_code = status_code + self.message = message diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/result.py b/litellm/llms/anthropic/experimental_pass_through/context_management/result.py new file mode 100644 index 00000000000..28641a90993 --- /dev/null +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/result.py @@ -0,0 +1,24 @@ +"""``PolyfillResult`` — the shape returned by the context-management dispatcher. + +Threaded from the dispatcher through ``async_anthropic_messages_handler`` into +the adapter so it can prepend the ``compaction`` block to the response and +attach ``iterations`` to ``usage``. +""" + +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional, Union + +from litellm.types.llms.anthropic import ( + AppliedEdit, + CompactionBlock, + UsageIteration, +) + + +@dataclass +class PolyfillResult: + messages: List[Dict[str, Any]] + system: Optional[Union[str, List[Dict[str, Any]]]] + applied_edits: List[AppliedEdit] = field(default_factory=list) + compaction_block: Optional[CompactionBlock] = None + iterations_usage: Optional[List[UsageIteration]] = None diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 2551353a049..2210caa5a79 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -459,40 +459,12 @@ def anthropic_messages_handler( **_shared_kwargs ) - # In-gateway context_management polyfill on the chat-completions adapter - # (native on Anthropic/Responses paths). Skipped when drop_params is on. - context_management_spec = _shared_kwargs.pop("context_management", None) - _request_drop_params = _shared_kwargs.get("drop_params") - _drop_params = ( - _request_drop_params - if _request_drop_params is not None - else litellm.drop_params - ) - polyfill_applied_edits: Optional[List[AppliedEdit]] = None - if context_management_spec and not _drop_params: - from litellm.llms.anthropic.experimental_pass_through.context_management import ( - apply_context_management, - ) - - try: - edited_messages, polyfill_applied_edits = apply_context_management( - model=model, - messages=_shared_kwargs["messages"], - tools=_shared_kwargs.get("tools"), - system=_shared_kwargs.get("system"), - context_management_spec=context_management_spec, - ) - _shared_kwargs["messages"] = edited_messages - except Exception as e: - verbose_logger.exception( - "context_management polyfill: skipping edits due to error: %s", - e, - ) - polyfill_applied_edits = None - + # The in-gateway context_management polyfill runs inside + # ``async_anthropic_messages_handler`` so it can ``await`` the + # summarization model for ``compact_20260112``. ``context_management`` + # is passed through as a regular kwarg. return ( LiteLLMMessagesToCompletionTransformationHandler.anthropic_messages_handler( - _polyfill_applied_edits=polyfill_applied_edits, **_shared_kwargs, ) ) diff --git a/litellm/proxy/anthropic_endpoints/endpoints.py b/litellm/proxy/anthropic_endpoints/endpoints.py index 69d69354fd1..79e2648744b 100644 --- a/litellm/proxy/anthropic_endpoints/endpoints.py +++ b/litellm/proxy/anthropic_endpoints/endpoints.py @@ -3,10 +3,14 @@ Unified /v1/messages endpoint - (Anthropic Spec) """ from fastapi import APIRouter, Depends, HTTPException, Request, Response +from fastapi.responses import JSONResponse from litellm._logging import verbose_proxy_logger from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping from litellm.integrations.custom_guardrail import ModifyResponseException +from litellm.llms.anthropic.experimental_pass_through.context_management import ( + AnthropicContextManagementError, +) from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.common_request_processing import ( @@ -114,6 +118,13 @@ async def anthropic_response( # noqa: PLR0915 ) return _anthropic_response + except AnthropicContextManagementError as e: + body = AnthropicExceptionMapping.transform_to_anthropic_error( + status_code=e.status_code, + raw_message=e.message, + request_id=request.headers.get("x-request-id"), + ) + return JSONResponse(status_code=e.status_code, content=body) except Exception as e: await proxy_logging_obj.post_call_failure_hook( user_api_key_dict=user_api_key_dict, original_exception=e, request_data=data diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index 4655951ae24..2fb881cb003 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -521,6 +521,11 @@ class AppliedEdit(TypedDict, total=False): cleared_input_tokens: int cleared_tool_uses: int cleared_thinking_turns: int + # compact_20260112 fields + summary_input_tokens: int + summary_output_tokens: int + error: str + warnings: List[str] class ContextManagementResponse(TypedDict, total=False): @@ -529,6 +534,21 @@ class ContextManagementResponse(TypedDict, total=False): applied_edits: List[AppliedEdit] +class CompactionBlock(TypedDict, total=False): + """Synthesized ``compaction`` content block (compact_20260112).""" + + type: Literal["compaction"] + content: Optional[str] + + +class UsageIteration(TypedDict, total=False): + """One sampling iteration's token usage (compact_20260112).""" + + type: Literal["compaction", "message"] + input_tokens: int + output_tokens: int + + class MessageBlockDelta(TypedDict): """ Anthropic diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index 44530fecebd..74e1e17e6d7 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -2472,3 +2472,172 @@ def test_translate_anthropic_tool_choice_none(): result = adapter.translate_anthropic_tool_choice_to_openai({"type": "none"}) assert result == "none" + + +# --------------------------------------------------------------------------- +# PolyfillResult integration tests +# --------------------------------------------------------------------------- + + +def _make_simple_openai_response( + text: str = "Hello", prompt_tokens: int = 10, completion_tokens: int = 5 +) -> ModelResponse: + return ModelResponse( + id="resp_polyfill_test", + model="gpt-4o", + choices=[ + Choices( + finish_reason="stop", + message=Message(role="assistant", content=text), + ) + ], + usage=Usage(prompt_tokens=prompt_tokens, completion_tokens=completion_tokens), + ) + + +def test_translate_openai_response_to_anthropic_with_polyfill_compaction_block(): + """compaction_block from PolyfillResult must be prepended to content at index 0.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.result import ( + PolyfillResult, + ) + + compaction_block = {"type": "compaction", "content": "Summary of prior turns."} + polyfill = PolyfillResult( + messages=[], + system=None, + applied_edits=[{"type": "compact_20260112"}], + compaction_block=compaction_block, + iterations_usage=None, + ) + response = _make_simple_openai_response(text="Hello after compaction.") + adapter = LiteLLMAnthropicMessagesAdapter() + result = adapter.translate_openai_response_to_anthropic( + response=response, polyfill_result=polyfill + ) + + content = result.get("content") + assert content is not None + assert content[0]["type"] == "compaction" + assert content[0]["content"] == "Summary of prior turns." + assert content[1]["type"] == "text" + assert content[1]["text"] == "Hello after compaction." + + # applied_edits must surface on context_management + cm = result.get("context_management") + assert cm is not None + assert cm["applied_edits"][0]["type"] == "compact_20260112" + + +def test_translate_openai_response_to_anthropic_with_polyfill_iterations_usage(): + """iterations_usage from PolyfillResult must produce usage['iterations'] with a message entry.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.result import ( + PolyfillResult, + ) + + polyfill = PolyfillResult( + messages=[], + system=None, + applied_edits=[{"type": "compact_20260112"}], + compaction_block=None, + iterations_usage=[ + {"type": "compaction", "input_tokens": 200, "output_tokens": 50}, + ], + ) + response = _make_simple_openai_response(prompt_tokens=100, completion_tokens=30) + adapter = LiteLLMAnthropicMessagesAdapter() + result = adapter.translate_openai_response_to_anthropic( + response=response, polyfill_result=polyfill + ) + + usage = result.get("usage") + assert usage is not None + iterations = usage.get("iterations") + assert iterations is not None + assert len(iterations) == 2 + assert iterations[0] == { + "type": "compaction", + "input_tokens": 200, + "output_tokens": 50, + } + assert iterations[1]["type"] == "message" + assert iterations[1]["input_tokens"] == 100 + assert iterations[1]["output_tokens"] == 30 + + # Top-level tokens must still reflect the message iteration + assert usage["input_tokens"] == 100 + assert usage["output_tokens"] == 30 + + +def test_translate_openai_response_to_anthropic_no_polyfill_no_change(): + """Without a PolyfillResult the response must be unchanged (no compaction, no iterations).""" + response = _make_simple_openai_response() + adapter = LiteLLMAnthropicMessagesAdapter() + result = adapter.translate_openai_response_to_anthropic(response=response) + + content = result.get("content") + assert content is not None + assert content[0]["type"] == "text" + + usage = result.get("usage") + assert usage is not None + assert "iterations" not in usage + + +def test_translate_openai_response_to_anthropic_with_polyfill_both_compaction_and_iterations(): + """Full summary path: compaction_block and iterations_usage both present simultaneously.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.result import ( + PolyfillResult, + ) + + compaction_block = { + "type": "compaction", + "content": "Summary of a long conversation.", + } + polyfill = PolyfillResult( + messages=[], + system=None, + applied_edits=[{"type": "compact_20260112"}], + compaction_block=compaction_block, + iterations_usage=[ + {"type": "compaction", "input_tokens": 300, "output_tokens": 75}, + ], + ) + response = _make_simple_openai_response( + text="After compaction.", prompt_tokens=120, completion_tokens=40 + ) + adapter = LiteLLMAnthropicMessagesAdapter() + result = adapter.translate_openai_response_to_anthropic( + response=response, polyfill_result=polyfill + ) + + # compaction block must come first + content = result.get("content") + assert content is not None + assert content[0]["type"] == "compaction" + assert content[0]["content"] == "Summary of a long conversation." + assert content[1]["type"] == "text" + assert content[1]["text"] == "After compaction." + + # iterations: compaction entry + message entry + usage = result.get("usage") + assert usage is not None + iterations = usage.get("iterations") + assert iterations is not None + assert len(iterations) == 2 + assert iterations[0] == { + "type": "compaction", + "input_tokens": 300, + "output_tokens": 75, + } + assert iterations[1]["type"] == "message" + assert iterations[1]["input_tokens"] == 120 + assert iterations[1]["output_tokens"] == 40 + + # top-level tokens match the message iteration + assert usage["input_tokens"] == 120 + assert usage["output_tokens"] == 40 + + # context_management applied_edits must surface + cm = result.get("context_management") + assert cm is not None + assert cm["applied_edits"][0]["type"] == "compact_20260112" diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py new file mode 100644 index 00000000000..96e5f23fc8c --- /dev/null +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -0,0 +1,862 @@ +""" +Unit tests for the compact_20260112 polyfill editor. + +Coverage: +- trigger.value < 50k → AnthropicContextManagementError(400) +- opt-in gate (no summary model) → summary_model_not_configured +- slice-only path (existing compaction block, under threshold) +- full summary path (over threshold, summary fires) +- summary call raises → summary_call_failed +- summary response missing tags → summary_extraction_failed +- pause_after_compaction: true → pause_after_compaction_ignored warning, proceeds +- custom instructions → default prompt is not used even when tools present +""" + +from typing import Any, Dict, List +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +from litellm.llms.anthropic.experimental_pass_through.context_management import ( + AnthropicContextManagementError, + apply_context_management, +) +from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _augment_system_with_summary, + _extract_summary_text, + _slice_around_compaction_block, + _strip_compaction_blocks, + apply_compact_20260112, +) + +MODEL = "openai/gpt-4o" + +_EDIT_SPEC_DEFAULT: Dict[str, Any] = {"type": "compact_20260112"} + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _simple_messages() -> List[Dict[str, Any]]: + return [ + {"role": "user", "content": "Hello"}, + {"role": "assistant", "content": [{"type": "text", "text": "Hi there"}]}, + {"role": "user", "content": "What is 2+2?"}, + ] + + +def _messages_with_compaction(summary: str = "prev summary") -> List[Dict[str, Any]]: + """History that already has a compaction block in an assistant turn.""" + return [ + {"role": "user", "content": "older question"}, + { + "role": "assistant", + "content": [{"type": "compaction", "content": summary}], + }, + {"role": "user", "content": "newer question"}, + {"role": "assistant", "content": [{"type": "text", "text": "newer reply"}]}, + {"role": "user", "content": "latest question"}, + ] + + +def _make_mock_response( + content: str, + prompt_tokens: int = 50, + completion_tokens: int = 100, +) -> MagicMock: + response = MagicMock() + choice = MagicMock() + message = MagicMock() + message.content = content + choice.message = message + response.choices = [choice] + usage = MagicMock() + usage.prompt_tokens = prompt_tokens + usage.completion_tokens = completion_tokens + response.usage = usage + return response + + +# --------------------------------------------------------------------------- +# Unit: helper functions +# --------------------------------------------------------------------------- + + +def test_slice_around_compaction_block_found(): + messages = _messages_with_compaction("my summary") + sliced, block = _slice_around_compaction_block(messages) + assert block is not None + assert block["type"] == "compaction" + assert block["content"] == "my summary" + # Sliced list starts at the assistant turn containing the compaction block + assert sliced[0]["role"] == "assistant" + assert len(sliced) == 4 # assistant(compaction), user, assistant, user + + +def test_slice_around_compaction_block_not_found(): + messages = _simple_messages() + sliced, block = _slice_around_compaction_block(messages) + assert block is None + assert sliced is messages # same object, no copy + + +def test_strip_compaction_blocks_removes_block(): + messages = [ + { + "role": "assistant", + "content": [ + {"type": "compaction", "content": "summary"}, + {"type": "text", "text": "hello"}, + ], + } + ] + stripped = _strip_compaction_blocks(messages) + assert len(stripped) == 1 + content = stripped[0]["content"] + assert all(b["type"] != "compaction" for b in content) + assert len(content) == 1 + assert content[0]["type"] == "text" + + +def test_strip_compaction_blocks_drops_compaction_only_turn(): + messages = [ + {"role": "user", "content": "hi"}, + { + "role": "assistant", + "content": [{"type": "compaction", "content": "summary"}], + }, + {"role": "user", "content": "bye"}, + ] + stripped = _strip_compaction_blocks(messages) + assert len(stripped) == 2 + assert stripped[0]["role"] == "user" + assert stripped[1]["role"] == "user" + + +def test_augment_system_with_summary_none_system(): + result = _augment_system_with_summary(None, "my summary") + assert isinstance(result, str) + assert "my summary" in result + + +def test_augment_system_with_summary_string_system(): + result = _augment_system_with_summary("You are helpful.", "my summary") + assert isinstance(result, str) + assert result.startswith("Previous conversation summary:") + assert "my summary" in result + assert "You are helpful." in result + + +def test_augment_system_with_summary_list_system(): + system = [{"type": "text", "text": "existing system"}] + result = _augment_system_with_summary(system, "my summary") + assert isinstance(result, list) + assert result[0]["type"] == "text" + text = result[0]["text"] + assert "my summary" in text + assert "existing system" in text + + +def test_extract_summary_text_found(): + raw = "Here is the summary:\nKey points from chat\nDone." + assert _extract_summary_text(raw) == "Key points from chat" + + +def test_extract_summary_text_missing_tags(): + assert _extract_summary_text("No tags here") is None + + +def test_extract_summary_text_none(): + assert _extract_summary_text(None) is None + + +def test_extract_summary_text_case_insensitive(): + raw = "uppercase tags" + assert _extract_summary_text(raw) == "uppercase tags" + + +# --------------------------------------------------------------------------- +# Editor: validation +# --------------------------------------------------------------------------- + + +async def test_trigger_below_minimum_raises(): + with pytest.raises(AnthropicContextManagementError) as exc_info: + await apply_compact_20260112( + model=MODEL, + messages=_simple_messages(), + tools=None, + system=None, + edit_spec={ + "type": "compact_20260112", + "trigger": {"type": "input_tokens", "value": 10_000}, + }, + ) + assert exc_info.value.status_code == 400 + assert "50000" in exc_info.value.message + + +async def test_trigger_at_minimum_does_not_raise(): + """Exactly 50 000 is allowed — only strictly less than 50k is rejected.""" + with patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value=None, + ): + result = await apply_compact_20260112( + model=MODEL, + messages=_simple_messages(), + tools=None, + system=None, + edit_spec={ + "type": "compact_20260112", + "trigger": {"type": "input_tokens", "value": 50_000}, + }, + ) + # Reached opt-in gate (no summary model); no error raised from trigger check + assert result.applied_edits[0]["error"] == "summary_model_not_configured" + + +# --------------------------------------------------------------------------- +# Editor: opt-in gate +# --------------------------------------------------------------------------- + + +async def test_opt_in_gating_no_summary_model_configured(): + messages = _simple_messages() + with patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value=None, + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system="system prompt", + edit_spec=_EDIT_SPEC_DEFAULT, + ) + assert result.applied_edits[0]["error"] == "summary_model_not_configured" + assert result.messages == messages + assert result.system == "system prompt" + assert result.compaction_block is None + assert result.iterations_usage is None + + +# --------------------------------------------------------------------------- +# Editor: slice-only path +# --------------------------------------------------------------------------- + + +async def test_slice_only_path_with_existing_compaction_block(): + """Phase A slices; Phase B token count is below threshold; no summary call.""" + messages = _messages_with_compaction("prior summary text") + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=500), # well under threshold + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + # System should have the prior summary prefixed + assert result.system is not None + assert "prior summary text" in str(result.system) + + # No new compaction block; no iterations_usage + assert result.compaction_block is None + assert result.iterations_usage is None + + # No compaction block in downstream messages + for msg in result.messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + assert block.get("type") != "compaction" + + +async def test_slice_only_no_compaction_block_under_threshold(): + """No prior compaction block, and token count is below threshold — pure pass-through.""" + messages = _simple_messages() + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=500), + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + assert result.messages == messages + assert result.compaction_block is None + assert result.iterations_usage is None + assert not result.applied_edits[0].get("error") + + +# --------------------------------------------------------------------------- +# Editor: full summary path +# --------------------------------------------------------------------------- + + +async def test_full_summary_path(): + """Over threshold: summary call fires, compaction_block and iterations_usage returned.""" + messages = _simple_messages() + mock_response = _make_mock_response( + "Condensed history", prompt_tokens=200, completion_tokens=50 + ) + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), # over 150k threshold + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + new_callable=AsyncMock, + return_value=mock_response, + ), + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + assert result.compaction_block is not None + assert result.compaction_block["type"] == "compaction" + assert result.compaction_block["content"] == "Condensed history" + + assert result.iterations_usage is not None + assert len(result.iterations_usage) == 1 + assert result.iterations_usage[0]["type"] == "compaction" + assert result.iterations_usage[0]["input_tokens"] == 200 + assert result.iterations_usage[0]["output_tokens"] == 50 + + # System must have summary prefixed + assert "Condensed history" in str(result.system) + + # applied_edits should have usage fields + edit = result.applied_edits[0] + assert edit["type"] == "compact_20260112" + assert edit.get("summary_input_tokens") == 200 + assert edit.get("summary_output_tokens") == 50 + + # Downstream messages must not contain a compaction block + for msg in result.messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + assert block.get("type") != "compaction" + + +async def test_full_summary_path_uses_router_when_available(): + """When llm_router is provided, its acompletion method is called instead of litellm.""" + messages = _simple_messages() + mock_response = _make_mock_response("Router summary") + mock_router = MagicMock() + mock_router.acompletion = AsyncMock(return_value=mock_response) + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="my-summary-model", + ), + patch("litellm.token_counter", return_value=200_000), + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + llm_router=mock_router, + ) + + mock_router.acompletion.assert_called_once() + call_kwargs = mock_router.acompletion.call_args.kwargs + assert call_kwargs["model"] == "my-summary-model" + + assert result.compaction_block is not None + assert result.compaction_block["content"] == "Router summary" + + +async def test_metadata_propagated_to_summary_call(): + """Auth metadata from the parent request is forwarded to the summary call.""" + messages = _simple_messages() + mock_response = _make_mock_response("Summary") + parent_metadata = { + "user_api_key": "sk-test", + "user_api_key_team_id": "team-123", + "user_api_key_user_id": "user-456", + "litellm_call_id": "call-789", + "should_not_propagate": "secret", + } + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + new_callable=AsyncMock, + return_value=mock_response, + ) as mock_call, + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + metadata=parent_metadata, + ) + + call_kwargs = mock_call.call_args.kwargs + propagated = call_kwargs["metadata"] + assert propagated["user_api_key"] == "sk-test" + assert propagated["user_api_key_team_id"] == "team-123" + assert "should_not_propagate" not in propagated + + +# --------------------------------------------------------------------------- +# Editor: error paths +# --------------------------------------------------------------------------- + + +async def test_summary_call_failed(): + """When the summary model raises, applied_edits[0].error == 'summary_call_failed'.""" + messages = _simple_messages() + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + new_callable=AsyncMock, + side_effect=RuntimeError("network error"), + ), + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + assert result.applied_edits[0]["error"] == "summary_call_failed" + assert result.compaction_block is None + assert result.iterations_usage is None + # Messages passed through (at minimum sliced, no compaction blocks) + for msg in result.messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + assert block.get("type") != "compaction" + + +async def test_summary_extraction_failed_no_tags(): + """When summary response has no tags, applied_edits[0].error == 'summary_extraction_failed'.""" + messages = _simple_messages() + mock_response = _make_mock_response("I cannot summarize that.") + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + new_callable=AsyncMock, + return_value=mock_response, + ), + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + assert result.applied_edits[0]["error"] == "summary_extraction_failed" + assert result.compaction_block is None + assert result.iterations_usage is None + + +# --------------------------------------------------------------------------- +# Editor: warnings +# --------------------------------------------------------------------------- + + +async def test_pause_after_compaction_ignored_warning(): + """pause_after_compaction: true → warning recorded, request proceeds normally.""" + messages = _simple_messages() + with patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value=None, + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec={ + "type": "compact_20260112", + "pause_after_compaction": True, + }, + ) + + edit = result.applied_edits[0] + assert "pause_after_compaction_ignored" in (edit.get("warnings") or []) + # Request still proceeds (here it hits opt-in gate because no model configured) + assert edit.get("error") == "summary_model_not_configured" + + +async def test_unsupported_trigger_type_falls_back_to_default(): + messages = _simple_messages() + with patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value=None, + ): + result = await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec={ + "type": "compact_20260112", + "trigger": {"type": "output_tokens", "value": 200_000}, + }, + ) + + edit = result.applied_edits[0] + warnings = edit.get("warnings") or [] + assert any("unsupported_trigger_type" in w for w in warnings) + + +# --------------------------------------------------------------------------- +# Editor: custom instructions +# --------------------------------------------------------------------------- + + +async def test_custom_instructions_used_verbatim(): + """Custom instructions are used as-is; the default prompt is NOT appended.""" + messages = _simple_messages() + tools = [{"name": "search", "description": "Search tool"}] + mock_response = _make_mock_response("Custom summary") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=tools, + system=None, + edit_spec={ + "type": "compact_20260112", + "instructions": "Summarize everything briefly.", + }, + ) + + assert len(captured_calls) == 1 + summary_messages = captured_calls[0]["summary_messages"] + # The last message should be the custom instruction prompt + last_msg = summary_messages[-1] + assert last_msg["role"] == "user" + assert last_msg["content"] == "Summarize everything briefly." + # The "do not call tools" suffix should NOT be in the prompt since custom was set + assert "tool" not in last_msg["content"].lower() + + +async def test_default_instructions_appended_with_no_tool_suffix_when_no_tools(): + """Without tools, default prompt is used but the no-tool-calls suffix is absent.""" + messages = _simple_messages() + mock_response = _make_mock_response("Default summary") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + prompt = captured_calls[0]["summary_messages"][-1]["content"] + # Should not contain the no-tool-calls guidance + assert "do not call" not in prompt.lower() + + +async def test_default_instructions_with_tools_appends_no_tool_suffix(): + """With tools and no custom instructions, the no-tool-calls suffix is appended.""" + messages = _simple_messages() + tools = [{"name": "search"}] + mock_response = _make_mock_response("Tool-aware summary") + + captured_calls: list = [] + + async def _fake_call_summary_model(**kwargs): + captured_calls.append(kwargs) + return mock_response + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value="claude-haiku-4-5", + ), + patch("litellm.token_counter", return_value=200_000), + patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._call_summary_model", + side_effect=_fake_call_summary_model, + ), + ): + await apply_compact_20260112( + model=MODEL, + messages=messages, + tools=tools, + system=None, + edit_spec=_EDIT_SPEC_DEFAULT, + ) + + prompt = captured_calls[0]["summary_messages"][-1]["content"] + assert "tool" in prompt.lower() + + +# --------------------------------------------------------------------------- +# Dispatcher integration: compact_20260112 via apply_context_management +# --------------------------------------------------------------------------- + + +async def test_dispatcher_routes_compact_edit(): + """compact_20260112 in the dispatcher resolves to opt-in gate when no model set.""" + messages = _simple_messages() + with patch( + "litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting", + return_value=None, + ): + result = await apply_context_management( + model=MODEL, + messages=messages, + tools=None, + system=None, + context_management_spec={"edits": [{"type": "compact_20260112"}]}, + ) + + assert len(result.applied_edits) == 1 + assert result.applied_edits[0]["type"] == "compact_20260112" + assert result.applied_edits[0].get("error") == "summary_model_not_configured" + + +async def test_dispatcher_trigger_below_minimum_raises_through(): + """AnthropicContextManagementError from the editor bubbles up through the dispatcher.""" + with pytest.raises(AnthropicContextManagementError): + await apply_context_management( + model=MODEL, + messages=_simple_messages(), + tools=None, + system=None, + context_management_spec={ + "edits": [ + { + "type": "compact_20260112", + "trigger": {"type": "input_tokens", "value": 1_000}, + } + ] + }, + ) + + +# --------------------------------------------------------------------------- +# _run_polyfill_if_enabled: drop_params gate +# --------------------------------------------------------------------------- + + +async def test_run_polyfill_skipped_when_drop_params_true(): + """When drop_params=True the polyfill must be skipped (returns None).""" + from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( + _run_polyfill_if_enabled, + ) + + result = await _run_polyfill_if_enabled( + model=MODEL, + messages=_simple_messages(), + tools=None, + system=None, + context_management_spec={"edits": [{"type": "compact_20260112"}]}, + metadata={}, + drop_params=True, + llm_router=None, + ) + assert result is None + + +async def test_run_polyfill_skipped_when_spec_empty(): + """Empty context_management_spec must also return None (no polyfill work).""" + from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( + _run_polyfill_if_enabled, + ) + + result = await _run_polyfill_if_enabled( + model=MODEL, + messages=_simple_messages(), + tools=None, + system=None, + context_management_spec=None, + metadata={}, + drop_params=False, + llm_router=None, + ) + assert result is None + + +# --------------------------------------------------------------------------- +# Endpoint error format: AnthropicContextManagementError → Anthropic 400 body +# --------------------------------------------------------------------------- + + +def test_anthropic_context_management_error_format(): + """AnthropicContextManagementError must produce an Anthropic-format body via + AnthropicExceptionMapping.transform_to_anthropic_error — the same path the + /v1/messages endpoint takes when it catches this exception.""" + from litellm.anthropic_interface.exceptions import AnthropicExceptionMapping + + body = AnthropicExceptionMapping.transform_to_anthropic_error( + status_code=400, + raw_message="trigger.value must be at least 50000 tokens", + request_id=None, + ) + + assert body["type"] == "error" + assert body["error"]["type"] == "invalid_request_error" + assert "50000" in body["error"]["message"] + + +def test_anthropic_context_management_error_attrs(): + """AnthropicContextManagementError carries status_code and message correctly.""" + err = AnthropicContextManagementError( + status_code=400, + message="trigger.value must be at least 50000 tokens", + ) + + assert err.status_code == 400 + assert "50000" in err.message + + +# --------------------------------------------------------------------------- +# Endpoint integration: /v1/messages → Anthropic 400 on context management error +# --------------------------------------------------------------------------- + + +def test_endpoint_returns_anthropic_400_on_context_management_error(): + """The /v1/messages endpoint must catch AnthropicContextManagementError and + return an Anthropic-format 400 JSONResponse — not a 500 ProxyException.""" + import sys + from unittest.mock import AsyncMock, MagicMock, patch + + from fastapi import FastAPI + from fastapi.testclient import TestClient + + from litellm.proxy.anthropic_endpoints.endpoints import router + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + + # Stub proxy_server to avoid apscheduler/heavy proxy deps imported lazily + # inside the route handler at request time. + mock_proxy_server = MagicMock() + mock_proxy_server.general_settings = {} + mock_proxy_server.llm_router = None + mock_proxy_server.proxy_config = MagicMock() + mock_proxy_server.proxy_logging_obj = MagicMock() + mock_proxy_server.user_api_base = None + mock_proxy_server.user_max_tokens = None + mock_proxy_server.user_model = None + mock_proxy_server.user_request_timeout = None + mock_proxy_server.user_temperature = None + mock_proxy_server.version = "test" + + with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}): + with patch( + "litellm.proxy.anthropic_endpoints.endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_cls: + mock_instance = MagicMock() + mock_instance.base_process_llm_request = AsyncMock( + side_effect=AnthropicContextManagementError( + status_code=400, + message="trigger.value must be at least 50000 tokens", + ) + ) + mock_cls.return_value = mock_instance + + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = lambda: MagicMock() + + client = TestClient(app, raise_server_exceptions=False) + response = client.post( + "/v1/messages", + json={ + "model": "gpt-4o", + "messages": [{"role": "user", "content": "hi"}], + }, + headers={"Authorization": "Bearer test-key"}, + ) + + assert response.status_code == 400 + body = response.json() + assert body["type"] == "error" + assert body["error"]["type"] == "invalid_request_error" + assert "50000" in body["error"]["message"] diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py index 924f79d0902..50c72cfe8d0 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_dispatcher.py @@ -43,9 +43,9 @@ def _history_with_two_tool_pairs(): ] -def test_unknown_edit_type_is_noop(): +async def test_unknown_edit_type_is_noop(): messages = _history_with_two_tool_pairs() - new_messages, applied = apply_context_management( + result = await apply_context_management( model=MODEL, messages=messages, tools=None, @@ -54,13 +54,13 @@ def test_unknown_edit_type_is_noop(): "edits": [{"type": "totally_not_a_real_edit_20999999"}] }, ) - assert applied == [] - assert new_messages == messages + assert result.applied_edits == [] + assert result.messages == messages -def test_known_edit_is_applied(): +async def test_known_edit_is_applied(): messages = _history_with_two_tool_pairs() - _, applied = apply_context_management( + result = await apply_context_management( model=MODEL, messages=messages, tools=None, @@ -75,14 +75,14 @@ def test_known_edit_is_applied(): ] }, ) - assert len(applied) == 1 - assert applied[0]["type"] == "clear_tool_uses_20250919" - assert applied[0]["cleared_tool_uses"] == 1 + assert len(result.applied_edits) == 1 + assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919" + assert result.applied_edits[0]["cleared_tool_uses"] == 1 -def test_mixed_known_unknown_only_known_applied(): +async def test_mixed_known_unknown_only_known_applied(): messages = _history_with_two_tool_pairs() - _, applied = apply_context_management( + result = await apply_context_management( model=MODEL, messages=messages, tools=None, @@ -99,33 +99,33 @@ def test_mixed_known_unknown_only_known_applied(): ] }, ) - assert len(applied) == 1 - assert applied[0]["type"] == "clear_tool_uses_20250919" + assert len(result.applied_edits) == 1 + assert result.applied_edits[0]["type"] == "clear_tool_uses_20250919" -def test_empty_or_missing_edits_list(): +async def test_empty_or_missing_edits_list(): messages = _history_with_two_tool_pairs() for spec in [{}, {"edits": None}, {"edits": []}, None]: - new_messages, applied = apply_context_management( + result = await apply_context_management( model=MODEL, messages=messages, tools=None, system=None, context_management_spec=spec, # type: ignore[arg-type] ) - assert applied == [] - assert new_messages == messages + assert result.applied_edits == [] + assert result.messages == messages -def test_malformed_edit_entries_are_skipped(): +async def test_malformed_edit_entries_are_skipped(): """Non-dict entries in `edits` list should be silently skipped.""" messages = _history_with_two_tool_pairs() - new_messages, applied = apply_context_management( + result = await apply_context_management( model=MODEL, messages=messages, tools=None, system=None, context_management_spec={"edits": ["not a dict", 42, None, {"type": None}]}, ) - assert applied == [] - assert new_messages == messages + assert result.applied_edits == [] + assert result.messages == messages