mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(compact): count the team model rate limit descriptor once
`_check_summary_model_rate_limit` called `_add_team_model_rate_limit_descriptor_from_metadata` itself on top of the copy `_create_rate_limit_descriptors` appends, handing the limiter one counter as two descriptors. #42516 consolidated the two `model_per_team` assembly sites for the pre-call path, where the repeat was charged and halved a team's configured per-model limit. It left this call site reaching for the helper separately, so the repeat moved here rather than disappearing. The repeat is harmless today only because this caller passes `read_only=True`, so the pair is read twice and nothing is incremented -- a property of the call site, not of the code it calls. Every counter consumer below charges a request once per descriptor it is given. Dropping the call also removes the `getattr` reaching for a private name and the guard that depended on it. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
2c9b0e00ac
commit
18e7ce3284
2 changed files with 55 additions and 9 deletions
|
|
@ -560,9 +560,6 @@ async def _check_summary_model_rate_limit(
|
|||
create_descriptors: Final[_CreateRateLimitDescriptors | None] = getattr(
|
||||
limiter, "_create_rate_limit_descriptors", None
|
||||
)
|
||||
add_team_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr(
|
||||
limiter, "_add_team_model_rate_limit_descriptor_from_metadata", None
|
||||
)
|
||||
add_project_descriptor: Final[_AddModelRateLimitDescriptor | None] = getattr(
|
||||
limiter, "_add_project_model_rate_limit_descriptor_from_metadata", None
|
||||
)
|
||||
|
|
@ -573,7 +570,6 @@ async def _check_summary_model_rate_limit(
|
|||
limiter is None
|
||||
or should_rate_limit_check is None
|
||||
or create_descriptors is None
|
||||
or add_team_descriptor is None
|
||||
or add_project_descriptor is None
|
||||
or create_org_descriptors is None
|
||||
):
|
||||
|
|
@ -589,11 +585,12 @@ async def _check_summary_model_rate_limit(
|
|||
tpm_limit_type=metadata.get("tpm_limit_type"),
|
||||
model_has_failures=False,
|
||||
)
|
||||
add_team_descriptor(
|
||||
user_api_key_dict=user_api_key_auth,
|
||||
requested_model=summary_model,
|
||||
descriptors=base_descriptors,
|
||||
)
|
||||
# The team per-model descriptor is appended by ``_create_rate_limit_descriptors``
|
||||
# itself; adding it again here repeats one (key, value) pair, and every
|
||||
# counter consumer charges the request once per descriptor it is given.
|
||||
# The repeat is currently harmless only because ``read_only=True`` below
|
||||
# reads the pair twice and increments nothing -- a property of this call
|
||||
# site rather than of the code it calls.
|
||||
add_project_descriptor(
|
||||
user_api_key_dict=user_api_key_auth,
|
||||
requested_model=summary_model,
|
||||
|
|
|
|||
|
|
@ -27,6 +27,7 @@ from litellm.llms.anthropic.pass_through.context_management import (
|
|||
)
|
||||
from litellm.llms.anthropic.pass_through.context_management.editors.compact import (
|
||||
_augment_system_with_summary,
|
||||
_check_summary_model_rate_limit,
|
||||
_extract_summary_text,
|
||||
_select_last_user_question,
|
||||
_slice_around_compaction_block,
|
||||
|
|
@ -2868,3 +2869,51 @@ async def test_threshold_check_counts_tokens_off_the_event_loop(monkeypatch):
|
|||
assert result.messages == messages
|
||||
assert result.compaction_block is None
|
||||
assert_loop_stayed_free(took, lags)
|
||||
|
||||
|
||||
async def test_summary_check_hands_the_team_model_counter_to_the_limiter_once():
|
||||
"""The team per-model descriptor reaches the read-only check exactly once.
|
||||
|
||||
``_create_rate_limit_descriptors`` appends it itself, so assembling it a
|
||||
second time here hands the limiter one counter twice. That is harmless
|
||||
while the check runs ``read_only=True``, but every counter consumer charges
|
||||
a request once per descriptor it is given, so a repeat is one refactor away
|
||||
from halving the team's configured per-model limit.
|
||||
"""
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
|
||||
_PROXY_MaxParallelRequestsHandler_v3,
|
||||
)
|
||||
from litellm.proxy.utils import InternalUsageCache, hash_token
|
||||
|
||||
summary_model = "claude-haiku-4-5"
|
||||
limiter = _PROXY_MaxParallelRequestsHandler_v3(
|
||||
internal_usage_cache=InternalUsageCache(DualCache())
|
||||
)
|
||||
captured: List[List[Dict[str, Any]]] = []
|
||||
|
||||
async def _capture(**kwargs):
|
||||
captured.append(list(kwargs.get("descriptors") or []))
|
||||
return {"overall_code": "OK"}
|
||||
|
||||
limiter.should_rate_limit = _capture
|
||||
|
||||
auth = UserAPIKeyAuth(
|
||||
api_key=hash_token("sk-team-model-dedup"),
|
||||
team_id="team-1",
|
||||
team_metadata={"model_rpm_limit": {summary_model: 2}},
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
_proxy_logging_like_the_live_proxy(limiter),
|
||||
):
|
||||
allowed = await _check_summary_model_rate_limit(auth, summary_model)
|
||||
|
||||
assert allowed is True
|
||||
assert captured, "the read-only rate-limit check never ran"
|
||||
team_descriptors = [d for d in captured[0] if d["key"] == "model_per_team"]
|
||||
assert len(team_descriptors) == 1, (
|
||||
f"expected one model_per_team descriptor, got {len(team_descriptors)}: {team_descriptors}"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue