From 09362b525da06fe8d48f565933fe392dad8cfae8 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Fri, 24 Apr 2026 20:04:52 -0700 Subject: [PATCH] fix: seed spend_counter_cache after DB reseed in _check_team_member_model_budget to prevent O(N) concurrent DB hits --- litellm/proxy/auth/auth_checks.py | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index ec005ce85e0..d48aceced64 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -3372,7 +3372,11 @@ async def _check_team_member_model_budget( if not models_to_check: return - from litellm.proxy.proxy_server import _reseed_spend_from_db, get_current_spend + from litellm.proxy.proxy_server import ( + _reseed_spend_from_db, + get_current_spend, + spend_counter_cache, + ) for model_str in models_to_check: model_budget_config = team_member_model_max_budget.get(model_str) @@ -3395,9 +3399,15 @@ async def _check_team_member_model_budget( fallback_spend=-1.0, ) if model_spend < 0: - # Pod restart or Redis flush — reseed from the authoritative DB row - # before evaluating the budget. Mirrors _init_and_increment_spend_counter. + # Cache cold (pod restart / Redis flush) — fetch authoritative value from DB. + # Seed spend_counter_cache immediately so concurrent auth calls for the same + # (user_id, team_id, model) slot use the cached value instead of each + # issuing another DB query (O(1) vs O(concurrent-requests) DB hits). model_spend = await _reseed_spend_from_db(counter_key) + if model_spend > 0: + await spend_counter_cache.async_increment_cache( + key=counter_key, value=model_spend + ) if model_spend >= max_budget: raise litellm.BudgetExceededError(