From e1423e82fc89c0e67a10a9a7b19374b14c687021 Mon Sep 17 00:00:00 2001 From: michaelxer Date: Mon, 8 Jun 2026 16:06:09 +0700 Subject: [PATCH] fix: exempt zero-cost models from internal-user max_budget check The _PROXY_MaxBudgetLimiter hook rejects requests when the internal user's spend exceeds max_budget, but it never checks whether the model has zero cost. This blocks self-hosted/on-prem models that add nothing to spend. Every other budget path (key, team, end-user) already exempts zero-cost models via _is_model_cost_zero in common_checks. This adds the same check to the max_budget_limiter hook. Fixes #29912 --- litellm/proxy/hooks/max_budget_limiter.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/litellm/proxy/hooks/max_budget_limiter.py b/litellm/proxy/hooks/max_budget_limiter.py index 769348a0b88..f3cb0cbbed7 100644 --- a/litellm/proxy/hooks/max_budget_limiter.py +++ b/litellm/proxy/hooks/max_budget_limiter.py @@ -69,6 +69,25 @@ class _PROXY_MaxBudgetLimiter(CustomLogger): resolved_model, llm_provider = resolve_llm_provider_for_rate_limit( data.get("model") if data else None ) + + # Zero-cost models (self-hosted / on-prem) add nothing to + # spend, so exempt them — matching the existing exemption in + # common_checks (user_api_key_auth.py) which already skips + # budget enforcement for zero-cost models on key, team, and + # end-user budget paths. + try: + from litellm.proxy.auth.auth_checks import _is_model_cost_zero + from litellm.proxy.proxy_server import llm_router + + if _is_model_cost_zero(resolved_model, llm_router): + verbose_proxy_logger.debug( + "MaxBudgetLimiter: Skipping budget check for zero-cost model: %s", + resolved_model, + ) + return + except ImportError: + pass + raise ProxyRateLimitError( detail="Max budget limit reached.", rate_limit_type=RateLimitType.BUDGET,