fix(proxy/hooks): populate llm_provider on budget/iterations 429s

Final batch of internal raise sites — the user/session-budget and
max-iterations hooks. Same pattern: resolve data['model'] once at
raise time, attach to ProxyHTTPRateLimitError so Prometheus and
observability callbacks can attribute the 429.

Hooks updated:
* max_budget_limiter (per-user max_budget exceeded)
* max_iterations_limiter (per-session agent iteration cap)
* max_budget_per_session_limiter (per-session dollar cap)

All three fall back to llm_provider='litellm_proxy' when data['model']
is missing or unparseable. Drops the now-unused HTTPException import
from each module.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:59:07 +00:00
parent 127d854ca4
commit c2af4271d8
No known key found for this signature in database
3 changed files with 33 additions and 7 deletions

View file

@ -5,6 +5,10 @@ from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
class _PROXY_MaxBudgetLimiter(CustomLogger):
@ -63,7 +67,15 @@ class _PROXY_MaxBudgetLimiter(CustomLogger):
# CHECK IF REQUEST ALLOWED
if curr_spend >= max_budget:
raise HTTPException(status_code=429, detail="Max budget limit reached.")
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model") if data else None
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail="Max budget limit reached.",
model=resolved_model,
llm_provider=llm_provider,
)
except HTTPException as e:
raise e
except Exception as e:

View file

@ -17,12 +17,14 @@ Follows the same pattern as max_iterations_limiter.py.
import os
from typing import TYPE_CHECKING, Any, Optional, Union
from fastapi import HTTPException
from litellm import DualCache
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
if TYPE_CHECKING:
from litellm.proxy.utils import InternalUsageCache as _InternalUsageCache
@ -112,13 +114,18 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
)
if current_spend >= max_budget:
raise HTTPException(
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model") if data else None
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail=(
f"Session budget exceeded for session {session_id}. "
f"Current spend: ${current_spend:.4f}, "
f"max_budget_per_session: ${max_budget:.2f}."
),
model=resolved_model,
llm_provider=llm_provider,
)
return None

View file

@ -13,12 +13,14 @@ Follows the same pattern as parallel_request_limiter_v3.py.
import os
from typing import TYPE_CHECKING, Any, Optional, Union
from fastapi import HTTPException
from litellm import DualCache
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
if TYPE_CHECKING:
from litellm.proxy.utils import InternalUsageCache as _InternalUsageCache
@ -116,12 +118,17 @@ class _PROXY_MaxIterationsHandler(CustomLogger):
current_count = await self._increment_and_get(cache_key)
if current_count > max_iterations:
raise HTTPException(
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model") if data else None
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail=(
f"Max iterations exceeded for session {session_id}. "
f"Current count: {current_count}, max_iterations: {max_iterations}."
),
model=resolved_model,
llm_provider=llm_provider,
)
verbose_proxy_logger.debug(