fix(compact): count the team model rate limit descriptor once

`_check_summary_model_rate_limit` called
`_add_team_model_rate_limit_descriptor_from_metadata` itself on top of the
copy `_create_rate_limit_descriptors` appends, handing the limiter one
counter as two descriptors.

#42516 consolidated the two `model_per_team` assembly sites for the
pre-call path, where the repeat was charged and halved a team's configured
per-model limit. It left this call site reaching for the helper separately,
so the repeat moved here rather than disappearing.

The repeat is harmless today only because this caller passes
`read_only=True`, so the pair is read twice and nothing is incremented --
a property of the call site, not of the code it calls. Every counter
consumer below charges a request once per descriptor it is given.

Dropping the call also removes the `getattr` reaching for a private name
and the guard that depended on it.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
DanBrima 2026-09-28 15:18:02 +03:00
parent 2c9b0e00ac
commit 18e7ce3284
2 changed files with 55 additions and 9 deletions

View file

@ -560,9 +560,6 @@ async def _check_summary_model_rate_limit(
create_descriptors: Final[_CreateRateLimitDescriptors | None] = getattr(
limiter, "_create_rate_limit_descriptors", None
)
add_team_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr(
limiter, "_add_team_model_rate_limit_descriptor_from_metadata", None
)
add_project_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr(
limiter, "_add_project_model_rate_limit_descriptor_from_metadata", None
)
@ -573,7 +570,6 @@ async def _check_summary_model_rate_limit(
limiter is None
or should_rate_limit_check is None
or create_descriptors is None
or add_team_descriptor is None
or add_project_descriptor is None
or create_org_descriptors is None
):
@ -589,11 +585,12 @@ async def _check_summary_model_rate_limit(
tpm_limit_type=metadata.get("tpm_limit_type"),
model_has_failures=False,
)
add_team_descriptor(
user_api_key_dict=user_api_key_auth,
requested_model=summary_model,
descriptors=base_descriptors,
)
# The team per-model descriptor is appended by ``_create_rate_limit_descriptors``
# itself; adding it again here repeats one (key, value) pair, and every
# counter consumer charges the request once per descriptor it is given.
# The repeat is currently harmless only because ``read_only=True`` below
# reads the pair twice and increments nothing -- a property of this call
# site rather than of the code it calls.
add_project_descriptor(
user_api_key_dict=user_api_key_auth,
requested_model=summary_model,

View file

@ -27,6 +27,7 @@ from litellm.llms.anthropic.pass_through.context_management import (
)
from litellm.llms.anthropic.pass_through.context_management.editors.compact import (
_augment_system_with_summary,
_check_summary_model_rate_limit,
_extract_summary_text,
_select_last_user_question,
_slice_around_compaction_block,
@ -2868,3 +2869,51 @@ async def test_threshold_check_counts_tokens_off_the_event_loop(monkeypatch):
assert result.messages == messages
assert result.compaction_block is None
assert_loop_stayed_free(took, lags)
async def test_summary_check_hands_the_team_model_counter_to_the_limiter_once():
"""The team per-model descriptor reaches the read-only check exactly once.
``_create_rate_limit_descriptors`` appends it itself, so assembling it a
second time here hands the limiter one counter twice. That is harmless
while the check runs ``read_only=True``, but every counter consumer charges
a request once per descriptor it is given, so a repeat is one refactor away
from halving the team's configured per-model limit.
"""
from litellm.caching.caching import DualCache
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
_PROXY_MaxParallelRequestsHandler_v3,
)
from litellm.proxy.utils import InternalUsageCache, hash_token
summary_model = "claude-haiku-4-5"
limiter = _PROXY_MaxParallelRequestsHandler_v3(
internal_usage_cache=InternalUsageCache(DualCache())
)
captured: List[List[Dict[str, Any]]] = []
async def _capture(**kwargs):
captured.append(list(kwargs.get("descriptors") or []))
return {"overall_code": "OK"}
limiter.should_rate_limit = _capture
auth = UserAPIKeyAuth(
api_key=hash_token("sk-team-model-dedup"),
team_id="team-1",
team_metadata={"model_rpm_limit": {summary_model: 2}},
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
_proxy_logging_like_the_live_proxy(limiter),
):
allowed = await _check_summary_model_rate_limit(auth, summary_model)
assert allowed is True
assert captured, "the read-only rate-limit check never ran"
team_descriptors = [d for d in captured[0] if d["key"] == "model_per_team"]
assert len(team_descriptors) == 1, (
f"expected one model_per_team descriptor, got {len(team_descriptors)}: {team_descriptors}"
)