mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix: seed spend_counter_cache after DB reseed in _check_team_member_model_budget to prevent O(N) concurrent DB hits
This commit is contained in:
parent
1fbb9a3bd3
commit
09362b525d
1 changed files with 13 additions and 3 deletions
|
|
@ -3372,7 +3372,11 @@ async def _check_team_member_model_budget(
|
|||
if not models_to_check:
|
||||
return
|
||||
|
||||
from litellm.proxy.proxy_server import _reseed_spend_from_db, get_current_spend
|
||||
from litellm.proxy.proxy_server import (
|
||||
_reseed_spend_from_db,
|
||||
get_current_spend,
|
||||
spend_counter_cache,
|
||||
)
|
||||
|
||||
for model_str in models_to_check:
|
||||
model_budget_config = team_member_model_max_budget.get(model_str)
|
||||
|
|
@ -3395,9 +3399,15 @@ async def _check_team_member_model_budget(
|
|||
fallback_spend=-1.0,
|
||||
)
|
||||
if model_spend < 0:
|
||||
# Pod restart or Redis flush — reseed from the authoritative DB row
|
||||
# before evaluating the budget. Mirrors _init_and_increment_spend_counter.
|
||||
# Cache cold (pod restart / Redis flush) — fetch authoritative value from DB.
|
||||
# Seed spend_counter_cache immediately so concurrent auth calls for the same
|
||||
# (user_id, team_id, model) slot use the cached value instead of each
|
||||
# issuing another DB query (O(1) vs O(concurrent-requests) DB hits).
|
||||
model_spend = await _reseed_spend_from_db(counter_key)
|
||||
if model_spend > 0:
|
||||
await spend_counter_cache.async_increment_cache(
|
||||
key=counter_key, value=model_spend
|
||||
)
|
||||
|
||||
if model_spend >= max_budget:
|
||||
raise litellm.BudgetExceededError(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue