fix(proxy): skip personal budget for zero-cost models

This commit is contained in:
冯基魁 2026-06-23 02:21:07 +08:00
parent ac7c2dc0d7
commit 68d19ef63a
2 changed files with 91 additions and 1 deletions

View file

@ -47,7 +47,17 @@ class _PROXY_MaxBudgetLimiter(CustomLogger):
):
return
from litellm.proxy.proxy_server import get_current_spend
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
from litellm.proxy.proxy_server import get_current_spend, llm_router
if data and _is_model_cost_zero(
model=data.get("model"), llm_router=llm_router
):
verbose_proxy_logger.info(
"MaxBudgetLimiter: skipping personal budget for zero-cost model=%s",
data.get("model"),
)
return
curr_spend = await get_current_spend(
counter_key=user_counter_key,

View file

@ -19,6 +19,7 @@ from fastapi import HTTPException
from litellm.caching.caching import DualCache
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.max_budget_limiter import _PROXY_MaxBudgetLimiter
from litellm.router import Router
def _make_user_api_key_auth(
@ -185,6 +186,85 @@ async def test_team_keys_skip_personal_budget():
mock_get_spend.assert_not_awaited()
@pytest.mark.asyncio
async def test_zero_cost_model_skips_personal_budget():
handler = _PROXY_MaxBudgetLimiter()
user_api_key_dict = _make_user_api_key_auth(
user_id="user-1",
user_max_budget=10.0,
user_spend=99.0,
)
router = Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "openai/free-model",
"api_key": "sk-test",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
}
]
)
with (
patch("litellm.proxy.proxy_server.llm_router", router),
patch(
"litellm.proxy.proxy_server.get_current_spend",
new=AsyncMock(return_value=99.0),
) as mock_get_spend,
):
result = await handler.async_pre_call_hook(
user_api_key_dict=user_api_key_dict,
cache=DualCache(),
data={"model": "free-model"},
call_type="completion",
)
assert result is None
mock_get_spend.assert_not_awaited()
@pytest.mark.asyncio
async def test_paid_model_still_enforces_personal_budget():
handler = _PROXY_MaxBudgetLimiter()
user_api_key_dict = _make_user_api_key_auth(
user_id="user-1",
user_max_budget=10.0,
user_spend=99.0,
)
router = Router(
model_list=[
{
"model_name": "paid-model",
"litellm_params": {
"model": "openai/gpt-4o-mini",
"api_key": "sk-test",
},
}
]
)
with (
patch("litellm.proxy.proxy_server.llm_router", router),
patch(
"litellm.proxy.proxy_server.get_current_spend",
new=AsyncMock(return_value=99.0),
),
):
with pytest.raises(HTTPException) as exc_info:
await handler.async_pre_call_hook(
user_api_key_dict=user_api_key_dict,
cache=DualCache(),
data={"model": "paid-model"},
call_type="completion",
)
assert exc_info.value.status_code == 429
assert "Max budget limit reached." in exc_info.value.detail
@pytest.mark.asyncio
async def test_no_max_budget_passes():
handler = _PROXY_MaxBudgetLimiter()