From 68d19ef63a0794bc44ed03e95cd9b6998915de01 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=86=AF=E5=9F=BA=E9=AD=81?= <1412414664@qq.com> Date: Tue, 23 Jun 2026 02:21:07 +0800 Subject: [PATCH] fix(proxy): skip personal budget for zero-cost models --- litellm/proxy/hooks/max_budget_limiter.py | 12 ++- .../proxy/hooks/test_max_budget_limiter.py | 80 +++++++++++++++++++ 2 files changed, 91 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/hooks/max_budget_limiter.py b/litellm/proxy/hooks/max_budget_limiter.py index 9a7e5117945..7bd8a226315 100644 --- a/litellm/proxy/hooks/max_budget_limiter.py +++ b/litellm/proxy/hooks/max_budget_limiter.py @@ -47,7 +47,17 @@ class _PROXY_MaxBudgetLimiter(CustomLogger): ): return - from litellm.proxy.proxy_server import get_current_spend + from litellm.proxy.auth.auth_checks import _is_model_cost_zero + from litellm.proxy.proxy_server import get_current_spend, llm_router + + if data and _is_model_cost_zero( + model=data.get("model"), llm_router=llm_router + ): + verbose_proxy_logger.info( + "MaxBudgetLimiter: skipping personal budget for zero-cost model=%s", + data.get("model"), + ) + return curr_spend = await get_current_spend( counter_key=user_counter_key, diff --git a/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py b/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py index 0074d7062b8..dd62f41a19a 100644 --- a/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py +++ b/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py @@ -19,6 +19,7 @@ from fastapi import HTTPException from litellm.caching.caching import DualCache from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.hooks.max_budget_limiter import _PROXY_MaxBudgetLimiter +from litellm.router import Router def _make_user_api_key_auth( @@ -185,6 +186,85 @@ async def test_team_keys_skip_personal_budget(): mock_get_spend.assert_not_awaited() +@pytest.mark.asyncio +async def test_zero_cost_model_skips_personal_budget(): + handler = _PROXY_MaxBudgetLimiter() + user_api_key_dict = _make_user_api_key_auth( + user_id="user-1", + user_max_budget=10.0, + user_spend=99.0, + ) + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "openai/free-model", + "api_key": "sk-test", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + } + ] + ) + + with ( + patch("litellm.proxy.proxy_server.llm_router", router), + patch( + "litellm.proxy.proxy_server.get_current_spend", + new=AsyncMock(return_value=99.0), + ) as mock_get_spend, + ): + result = await handler.async_pre_call_hook( + user_api_key_dict=user_api_key_dict, + cache=DualCache(), + data={"model": "free-model"}, + call_type="completion", + ) + + assert result is None + mock_get_spend.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_paid_model_still_enforces_personal_budget(): + handler = _PROXY_MaxBudgetLimiter() + user_api_key_dict = _make_user_api_key_auth( + user_id="user-1", + user_max_budget=10.0, + user_spend=99.0, + ) + router = Router( + model_list=[ + { + "model_name": "paid-model", + "litellm_params": { + "model": "openai/gpt-4o-mini", + "api_key": "sk-test", + }, + } + ] + ) + + with ( + patch("litellm.proxy.proxy_server.llm_router", router), + patch( + "litellm.proxy.proxy_server.get_current_spend", + new=AsyncMock(return_value=99.0), + ), + ): + with pytest.raises(HTTPException) as exc_info: + await handler.async_pre_call_hook( + user_api_key_dict=user_api_key_dict, + cache=DualCache(), + data={"model": "paid-model"}, + call_type="completion", + ) + + assert exc_info.value.status_code == 429 + assert "Max budget limit reached." in exc_info.value.detail + + @pytest.mark.asyncio async def test_no_max_budget_passes(): handler = _PROXY_MaxBudgetLimiter()