From 18e7ce32846c4181b8ba8003f30ed37aca5559df Mon Sep 17 00:00:00 2001 From: DanBrima Date: Mon, 28 Sep 2026 15:18:02 +0300 Subject: [PATCH] fix(compact): count the team model rate limit descriptor once `_check_summary_model_rate_limit` called `_add_team_model_rate_limit_descriptor_from_metadata` itself on top of the copy `_create_rate_limit_descriptors` appends, handing the limiter one counter as two descriptors. #42516 consolidated the two `model_per_team` assembly sites for the pre-call path, where the repeat was charged and halved a team's configured per-model limit. It left this call site reaching for the helper separately, so the repeat moved here rather than disappearing. The repeat is harmless today only because this caller passes `read_only=True`, so the pair is read twice and nothing is incremented -- a property of the call site, not of the code it calls. Every counter consumer below charges a request once per descriptor it is given. Dropping the call also removes the `getattr` reaching for a private name and the guard that depended on it. Co-Authored-By: Claude Opus 5 (1M context) --- .../context_management/editors/compact.py | 15 +++--- .../context_management/test_compact.py | 49 +++++++++++++++++++ 2 files changed, 55 insertions(+), 9 deletions(-) diff --git a/litellm/llms/anthropic/pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/pass_through/context_management/editors/compact.py index 62826865894..f24bfd2c70c 100644 --- a/litellm/llms/anthropic/pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/pass_through/context_management/editors/compact.py @@ -560,9 +560,6 @@ async def _check_summary_model_rate_limit( create_descriptors: Final[_CreateRateLimitDescriptors | None] = getattr( limiter, "_create_rate_limit_descriptors", None ) - add_team_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr( - limiter, "_add_team_model_rate_limit_descriptor_from_metadata", None - ) add_project_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr( limiter, "_add_project_model_rate_limit_descriptor_from_metadata", None ) @@ -573,7 +570,6 @@ async def _check_summary_model_rate_limit( limiter is None or should_rate_limit_check is None or create_descriptors is None - or add_team_descriptor is None or add_project_descriptor is None or create_org_descriptors is None ): @@ -589,11 +585,12 @@ async def _check_summary_model_rate_limit( tpm_limit_type=metadata.get("tpm_limit_type"), model_has_failures=False, ) - add_team_descriptor( - user_api_key_dict=user_api_key_auth, - requested_model=summary_model, - descriptors=base_descriptors, - ) + # The team per-model descriptor is appended by ``_create_rate_limit_descriptors`` + # itself; adding it again here repeats one (key, value) pair, and every + # counter consumer charges the request once per descriptor it is given. + # The repeat is currently harmless only because ``read_only=True`` below + # reads the pair twice and increments nothing -- a property of this call + # site rather than of the code it calls. add_project_descriptor( user_api_key_dict=user_api_key_auth, requested_model=summary_model, diff --git a/tests/unit/llms/anthropic/pass_through/context_management/test_compact.py b/tests/unit/llms/anthropic/pass_through/context_management/test_compact.py index bfba50fb368..992996bcfe8 100644 --- a/tests/unit/llms/anthropic/pass_through/context_management/test_compact.py +++ b/tests/unit/llms/anthropic/pass_through/context_management/test_compact.py @@ -27,6 +27,7 @@ from litellm.llms.anthropic.pass_through.context_management import ( ) from litellm.llms.anthropic.pass_through.context_management.editors.compact import ( _augment_system_with_summary, + _check_summary_model_rate_limit, _extract_summary_text, _select_last_user_question, _slice_around_compaction_block, @@ -2868,3 +2869,51 @@ async def test_threshold_check_counts_tokens_off_the_event_loop(monkeypatch): assert result.messages == messages assert result.compaction_block is None assert_loop_stayed_free(took, lags) + + +async def test_summary_check_hands_the_team_model_counter_to_the_limiter_once(): + """The team per-model descriptor reaches the read-only check exactly once. + + ``_create_rate_limit_descriptors`` appends it itself, so assembling it a + second time here hands the limiter one counter twice. That is harmless + while the check runs ``read_only=True``, but every counter consumer charges + a request once per descriptor it is given, so a repeat is one refactor away + from halving the team's configured per-model limit. + """ + from litellm.caching.caching import DualCache + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.hooks.parallel_request_limiter_v3 import ( + _PROXY_MaxParallelRequestsHandler_v3, + ) + from litellm.proxy.utils import InternalUsageCache, hash_token + + summary_model = "claude-haiku-4-5" + limiter = _PROXY_MaxParallelRequestsHandler_v3( + internal_usage_cache=InternalUsageCache(DualCache()) + ) + captured: List[List[Dict[str, Any]]] = [] + + async def _capture(**kwargs): + captured.append(list(kwargs.get("descriptors") or [])) + return {"overall_code": "OK"} + + limiter.should_rate_limit = _capture + + auth = UserAPIKeyAuth( + api_key=hash_token("sk-team-model-dedup"), + team_id="team-1", + team_metadata={"model_rpm_limit": {summary_model: 2}}, + ) + + with patch( + "litellm.proxy.proxy_server.proxy_logging_obj", + _proxy_logging_like_the_live_proxy(limiter), + ): + allowed = await _check_summary_model_rate_limit(auth, summary_model) + + assert allowed is True + assert captured, "the read-only rate-limit check never ran" + team_descriptors = [d for d in captured[0] if d["key"] == "model_per_team"] + assert len(team_descriptors) == 1, ( + f"expected one model_per_team descriptor, got {len(team_descriptors)}: {team_descriptors}" + )