From b15884357f821bad65ef5f82748ab7806bdeafa6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 28 May 2026 07:47:50 +0000 Subject: [PATCH] fix(compact_20260112): bound _call_summary_model with timeout A slow or unresponsive summary model previously hung the parent /v1/messages request with no escape hatch. Pass a 60s timeout on the litellm.acompletion / llm_router.acompletion subrequest; on timeout the existing summary_call_failed path forwards the request without compaction rather than blocking indefinitely. --- .../context_management/constants.py | 5 ++++ .../context_management/editors/compact.py | 6 +++++ .../context_management/test_compact.py | 27 +++++++++++++++++++ 3 files changed, 38 insertions(+) diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py index eb099407f00..ebbc182c427 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/constants.py @@ -18,6 +18,11 @@ COMPACT_MIN_TRIGGER_TOKENS = 50_000 # ``general_settings.context_management_summary_max_tokens``. COMPACT_SUMMARY_MAX_TOKENS = 4096 COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY = "context_management_summary_max_tokens" +# Wall-clock bound for the summary sub-call. Without this a slow or +# unresponsive summary model would hang the parent ``/v1/messages`` request +# with no escape hatch; on timeout the editor falls into the standard +# ``summary_call_failed`` path and forwards the request without compaction. +COMPACT_SUMMARY_TIMEOUT_SECONDS = 60.0 COMPACT_SUMMARY_MODEL_SETTING_KEY = "context_management_summary_model" COMPACT_SUMMARY_SYSTEM_PREFIX = "Previous conversation summary: " diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index 1fe8072d3d2..a85427a9180 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -33,6 +33,7 @@ from ..constants import ( COMPACT_SUMMARY_MAX_TOKENS_SETTING_KEY, COMPACT_SUMMARY_MODEL_SETTING_KEY, COMPACT_SUMMARY_SYSTEM_PREFIX, + COMPACT_SUMMARY_TIMEOUT_SECONDS, ) from ..errors import AnthropicContextManagementError from ..result import PolyfillResult @@ -569,10 +570,15 @@ async def _call_summary_model( # (see ``Router._common_checks_available_deployment``); without this the # summary subrequest could be routed to a deployment outside the caller's # permitted region. + # ``timeout`` bounds how long a slow/unresponsive summary model can stall + # the parent ``/v1/messages`` request. On timeout the caller catches the + # exception and surfaces ``applied_edits[0].error = "summary_call_failed"``, + # forwarding the request without compaction rather than hanging. call_kwargs: Dict[str, Any] = { "model": summary_model, "messages": summary_messages, "max_tokens": max_tokens, + "timeout": COMPACT_SUMMARY_TIMEOUT_SECONDS, "litellm_metadata": metadata, } if allowed_model_region is not None: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index 02b229ad0a1..580e3d37af8 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -1105,6 +1105,33 @@ def test_summary_max_tokens_setting_falls_back_for_invalid_values(): ), f"expected default for invalid override {bad!r}" +async def test_summary_call_sends_default_timeout(): + """``timeout`` is set on the summary call so a slow or unresponsive summary + model cannot hang the parent ``/v1/messages`` request indefinitely.""" + from litellm.llms.anthropic.experimental_pass_through.context_management.constants import ( + COMPACT_SUMMARY_TIMEOUT_SECONDS, + ) + from litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact import ( + _call_summary_model, + ) + + captured_kwargs: dict = {} + + class _FakeRouter: + async def acompletion(self, **kwargs): + captured_kwargs.update(kwargs) + return _make_mock_response("x") + + await _call_summary_model( + summary_model="claude-haiku-4-5", + summary_messages=[{"role": "user", "content": "hi"}], + metadata={}, + llm_router=_FakeRouter(), + ) + + assert captured_kwargs.get("timeout") == COMPACT_SUMMARY_TIMEOUT_SECONDS + + # --------------------------------------------------------------------------- # Editor: summary model key/team access gate # ---------------------------------------------------------------------------