diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py index 605e8cb6b99..eb099407f00 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py @@ -14,8 +14,10 @@ COMPACT_MIN_TRIGGER_TOKENS = 50_000 # Default ``max_tokens`` for the summary call. Required by providers like # Anthropic that reject requests without it; safely accepted by providers that # don't strictly require it. Chosen to comfortably fit a long structured -# summary. +# summary. Operators can override via +# ``general_settings.context_management_summary_max_tokens``. COMPACT_SUMMARY_MAX_TOKENS = 4096 +COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY = "context_management_summary_max_tokens" COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model" COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: " diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/clear_tool_uses.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/clear_tool_uses.py index a6c102db2b9..7b1c20ff522 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/clear_tool_uses.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/clear_tool_uses.py @@ -148,14 +148,18 @@ def apply_clear_tool_uses_20250919( edit_spec: Dict[str, Any], ) -> Tuple[List[Dict[str, Any]], Optional[AppliedEdit]]: """Apply clear_tool_uses; return (messages, AppliedEdit or None).""" - for ignored_knob in ("clear_at_least", "exclude_tools", "clear_tool_inputs"): - if ignored_knob in edit_spec: - verbose_logger.debug( - "context_management polyfill: ignoring '%s' on %s " - "(supported only on Anthropic-family forwarding path in v0)", - ignored_knob, - CLEAR_TOOL_USES_EDIT_TYPE, - ) + ignored_knobs = [ + knob + for knob in ("clear_at_least", "exclude_tools", "clear_tool_inputs") + if knob in edit_spec + ] + for ignored_knob in ignored_knobs: + verbose_logger.warning( + "context_management polyfill: ignoring '%s' on %s " + "(supported only on Anthropic-family forwarding path in v0)", + ignored_knob, + CLEAR_TOOL_USES_EDIT_TYPE, + ) trigger = edit_spec.get("trigger") or { "type": "input_tokens", @@ -201,4 +205,6 @@ def apply_clear_tool_uses_20250919( "cleared_tool_uses": cleared_count, "cleared_input_tokens": cleared_input_tokens, } + if ignored_knobs: + applied["warnings"] = [f"{knob}_ignored" for knob in ignored_knobs] return edited, applied diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index 76672999a98..1fe8072d3d2 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -30,6 +30,7 @@ from ..constants import ( COMPACT_MIN_TRIGGER_TOKENS, COMPACT_NO_TOOL_CALLS_SUFFIX, COMPACT_SUMMARY_MAX_TOKENS, + COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY, COMPACT_SUMMARY_MODEL_SETTING_KEY, COMPACT_SUMMARY_SYSTEM_PREFIX, ) @@ -65,6 +66,23 @@ def _read_summary_model_setting() -> Optional[str]: return value if isinstance(value, str) and value else None +def _read_summary_max_tokens_setting() -> int: + """Look up the configured summary ``max_tokens`` from proxy general_settings. + + Falls back to :data:`COMPACT_SUMMARY_MAX_TOKENS` when the setting is + missing or invalid (non-positive int, wrong type). Operators tune this + when the default doesn't fit their chosen summary model's output budget. + """ + try: + from litellm.proxy.proxy_server import general_settings + except Exception: + return COMPACT_SUMMARY_MAX_TOKENS + value = general_settings.get(COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY) + if isinstance(value, int) and value > 0: + return value + return COMPACT_SUMMARY_MAX_TOKENS + + def _check_summary_model_access( user_api_key_auth: Any, summary_model: str, @@ -526,6 +544,7 @@ async def _call_summary_model( metadata: Dict[str, Any], llm_router: Any, allowed_model_region: Optional[str] = None, + max_tokens: int = COMPACT_SUMMARY_MAX_TOKENS, ) -> Any: """Invoke the configured summary model. @@ -536,7 +555,10 @@ async def _call_summary_model( # ``max_tokens`` is required by providers like Anthropic and silently # accepted by providers that don't strictly require it (OpenAI etc.). # Setting a sensible default here means the feature works regardless of - # which model an admin configures as ``context_management_summary_model``. + # which model an admin configures as ``context_management_summary_model``; + # operators can override via ``context_management_summary_max_tokens`` in + # ``general_settings`` when the default doesn't fit the chosen model's + # output budget. # The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.) # must travel as ``litellm_metadata`` — that is the parameter the proxy's # post-call spend hooks read for budget attribution. The provider-level @@ -550,7 +572,7 @@ async def _call_summary_model( call_kwargs: Dict[str, Any] = { "model": summary_model, "messages": summary_messages, - "max_tokens": COMPACT_SUMMARY_MAX_TOKENS, + "max_tokens": max_tokens, "litellm_metadata": metadata, } if allowed_model_region is not None: @@ -771,6 +793,7 @@ async def apply_compact_20260112( # noqa: PLR0915 metadata=propagated_metadata, llm_router=llm_router, allowed_model_region=allowed_model_region, + max_tokens=_read_summary_max_tokens_setting(), ) except Exception as e: verbose_logger.warning("compact_20260112: summary call failed: %s", e) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_clear_tool_uses.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_clear_tool_uses.py index 9e450eb9082..09ac95ab16e 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_clear_tool_uses.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_clear_tool_uses.py @@ -224,6 +224,32 @@ def test_ignored_knobs_do_not_alter_behavior(): # Despite clear_tool_inputs=True, inputs are NOT cleared (knob ignored). assert applied is not None assert applied["cleared_tool_uses"] == 2 + # Ignored knobs surface as warnings on the AppliedEdit so operators can + # see what was dropped (the v0 polyfill silently dropping them at debug + # log level made misconfiguration invisible from the response). + assert set(applied.get("warnings", [])) == { + "clear_at_least_ignored", + "exclude_tools_ignored", + "clear_tool_inputs_ignored", + } + + +def test_no_ignored_knobs_omits_warnings_field(): + """When the caller doesn't pass any unsupported knobs, no ``warnings`` are added.""" + messages = _make_history(n_pairs=3) + _, applied = apply_clear_tool_uses_20250919( + model=MODEL, + messages=messages, + tools=None, + system=None, + edit_spec={ + "type": "clear_tool_uses_20250919", + "trigger": {"type": "tool_uses", "value": 0}, + "keep": {"type": "tool_uses", "value": 1}, + }, + ) + assert applied is not None + assert "warnings" not in applied def test_tool_result_list_content_shape_preserved(): diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index 2059038ad92..02b229ad0a1 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -1049,6 +1049,62 @@ async def test_summary_call_sends_default_max_tokens(): assert captured_kwargs.get("max_tokens") == COMPACT_SUMMARY_MAX_TOKENS +async def test_summary_call_honors_max_tokens_override(): + """Operators can override the default summary ``max_tokens`` via + ``general_settings.context_management_summary_max_tokens``.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _read_summary_max_tokens_setting, + ) + + captured_kwargs: dict = {} + + class _FakeRouter: + async def acompletion(self, **kwargs): + captured_kwargs.update(kwargs) + return _make_mock_response("x") + + with patch( + "litellm.proxy.proxy_server.general_settings", + {"context_management_summary_max_tokens": 8192}, + ): + assert _read_summary_max_tokens_setting() == 8192 + + from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _call_summary_model, + ) + + await _call_summary_model( + summary_model="claude-haiku-4-5", + summary_messages=[{"role": "user", "content": "hi"}], + metadata={}, + llm_router=_FakeRouter(), + max_tokens=_read_summary_max_tokens_setting(), + ) + + assert captured_kwargs.get("max_tokens") == 8192 + + +def test_summary_max_tokens_setting_falls_back_for_invalid_values(): + """Invalid override values (non-int, non-positive, missing) fall back to + the compiled default so a typo in ``general_settings`` doesn't break the + summary call.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.constants import ( + COMPACT_SUMMARY_MAX_TOKENS, + ) + from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _read_summary_max_tokens_setting, + ) + + for bad in ("4096", 0, -1, None, {"value": 1024}): + with patch( + "litellm.proxy.proxy_server.general_settings", + {"context_management_summary_max_tokens": bad}, + ): + assert ( + _read_summary_max_tokens_setting() == COMPACT_SUMMARY_MAX_TOKENS + ), f"expected default for invalid override {bad!r}" + + # --------------------------------------------------------------------------- # Editor: summary model key/team access gate # ---------------------------------------------------------------------------