diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 7b0e92d81ae..d1cb4ea7365 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -3106,6 +3106,12 @@ class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response ob ), ) budget_reservation: dict[str, Any] | None = Field(default=None, exclude=True) + # Whether this request is exempt from budget enforcement because the requested + # model is zero-cost. Decided once in user_api_key_auth() and published here so + # every enforcement point agrees, including the ones that run after auth. + # exclude=True keeps it out of serialised key responses; the default enforces + # the budget, so a path that never computed the exemption fails closed. + skip_budget_checks: bool = Field(default=False, exclude=True) matched_model_access_groups: list[str] | None = Field(default=None, exclude=True) budget_throttle_pct: float | None = Field(default=None, exclude=True) user: Any | None = None # Expanded user object when expand=user is used diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 18095aaafb4..fbee77d1876 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -436,7 +436,7 @@ def _is_model_cost_zero(model: str | list[str] | None, llm_router: Router | None # not from defaulted sparse auto-registration entries. # See: https://github.com/BerriAI/litellm/issues/24770 safe_name = str(model_name).replace("\n", "").replace("\r", "") - if not _is_cost_explicitly_configured(model_name, llm_router): + if not _group_declares_explicit_cost(model=model_name, llm_router=llm_router): verbose_proxy_logger.debug( "Model %s has zero cost but no explicit cost " "configuration in model_cost entry — treating as unknown " @@ -484,38 +484,18 @@ def _has_ptu_flat_cost(model: str, llm_router: "Router") -> bool: Such a deployment carries an explicit zero per-token price so the flat cost is not charged twice, which otherwise reads here as a free model and waives every budget check for it. + + Resolved through ``Router.get_model_list()`` so a model_group_alias pointing at a PTU group + is covered; scanning ``model_list`` by exact name never matches an alias, and the caller + treats a False here as "no flat cost to worry about". """ - for deployment in llm_router.model_list: - if deployment.get("model_name") != model: - continue + for deployment in llm_router.get_model_list(model_name=model) or (): model_info = deployment.get("model_info") or _NO_MODEL_INFO if model_info.get("ptu_count") is not None and model_info.get("cost_per_ptu_per_hour") is not None: return True return False -def _is_cost_explicitly_configured(model: str, llm_router: "Router") -> bool: - """ - Check if any deployment in the model group has cost fields explicitly - set in its litellm.model_cost entry. - - When Router._create_deployment() registers a model not in the global - cost map, it creates a sparse entry like {"id": ""} with no cost - fields. _get_model_info_helper() then defaults missing costs to 0. - This function detects that scenario by checking the raw model_cost entry. - """ - for deployment in llm_router.model_list: - if deployment.get("model_name") != model: - continue - model_id = deployment.get("model_info", {}).get("id") - if model_id is None: - continue - raw_entry = litellm.model_cost.get(model_id, {}) - if "input_cost_per_token" in raw_entry or "output_cost_per_token" in raw_entry: - return True - return False - - _EMPTY_COST_ENTRY: Final[Mapping[str, object]] = MappingProxyType({}) @@ -567,7 +547,7 @@ def _model_group_has_pricing(model: str, llm_router: "Router") -> bool: def _group_declares_explicit_cost(model: str, llm_router: "Router") -> bool: """ - Alias-aware counterpart to ``_is_cost_explicitly_configured``, which resolves the model group + Whether any deployment in the model group prices itself explicitly. Resolves the model group the same way ``_model_group_has_pricing`` does. A deployment that prices itself through its ``model_info`` block lands in the cost map under its deployment id rather than in its litellm_params, and reaching that entry through the router's own resolution keeps an alias diff --git a/litellm/proxy/auth/user_api_key_auth.py b/litellm/proxy/auth/user_api_key_auth.py index 20ab9904f46..4c8d10762af 100644 --- a/litellm/proxy/auth/user_api_key_auth.py +++ b/litellm/proxy/auth/user_api_key_auth.py @@ -1592,6 +1592,7 @@ async def _user_api_key_auth_builder( skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router) if skip_budget_checks: verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model) + valid_token.skip_budget_checks = skip_budget_checks # Fetch project object for JWT path if project_id is set _jwt_project_obj = None @@ -2019,6 +2020,7 @@ async def _user_api_key_auth_builder( skip_budget_checks = _is_model_cost_zero(model=model, llm_router=llm_router) if skip_budget_checks: verbose_proxy_logger.info("Skipping all budget checks for zero-cost model: %s", model) + valid_token.skip_budget_checks = skip_budget_checks # Check 3. Check if user is in their team budget if not skip_budget_checks and valid_token.team_member_spend is not None: @@ -2650,6 +2652,7 @@ async def _run_centralized_common_checks( request=request, llm_router=llm_router, ) + user_api_key_auth_obj.skip_budget_checks = skip_budget_checks # Pin the metadata variable name (litellm_metadata vs metadata) before # any tag merge runs. Without this, header tags from diff --git a/litellm/proxy/hooks/max_budget_limiter.py b/litellm/proxy/hooks/max_budget_limiter.py index eaf37b0bcf1..d5f3f296f18 100644 --- a/litellm/proxy/hooks/max_budget_limiter.py +++ b/litellm/proxy/hooks/max_budget_limiter.py @@ -32,6 +32,19 @@ class _PROXY_MaxBudgetLimiter(CustomLogger): if max_budget is None or user_id is None: return + # A zero-cost model is exempt from the personal budget, as it already is + # from the key, team and end-user budgets. The exemption is not recomputed + # here: user_api_key_auth() decided it and published it on the auth object, + # the same way budget_reservation below is handed in, so this hook cannot + # disagree with the checks that ran during auth. + if user_api_key_dict.skip_budget_checks: + verbose_proxy_logger.debug( + "MaxBudgetLimiter: user_id=%s is over budget but the request is exempt " + "(zero-cost model) - allowing", + user_id, + ) + return + from litellm.proxy.proxy_server import general_settings if ( diff --git a/litellm/router.py b/litellm/router.py index 8865543badd..8f57256fc47 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -9980,7 +9980,7 @@ class Router: A strategy-router alias is never the deployment actually called or billed, so custom pricing configured on it must not become a cost-map - price: an explicit zero would let ``_is_cost_explicitly_configured`` + price: an explicit zero would let ``_group_declares_explicit_cost`` treat the alias as a genuinely free model and waive budget checks for requests that route to (and bill as) a real deployment. """ diff --git a/litellm/utils.py b/litellm/utils.py index 1a77655a5a4..2889319d85d 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -3182,7 +3182,7 @@ def register_model( # = 0 when they are absent from the raw entry. Writing those zeros # back flips a sparse entry from "no cost keys" (priced via name) # to "cost keys = 0" (free), which makes - # ``_is_cost_explicitly_configured`` return True and silently + # ``_group_declares_explicit_cost`` return True and silently # disables budget enforcement on the next re-registration. _raw_entry = litellm.model_cost.get(model_cost_key) if _raw_entry is None: diff --git a/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py b/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py index bbe343bcede..9a87373f72b 100644 --- a/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py +++ b/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py @@ -211,3 +211,111 @@ class TestUnmappedModelBudgetEnforcement: result = _is_model_cost_zero(model="paid-model", llm_router=mock_router) assert result is False + + def test_model_group_alias_to_free_model_bypasses_budget(self): + """An explicitly free model reached through model_group_alias should + bypass budget, same as when it is called by its own name.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "openai/nonexistent-but-free-model", + "api_key": "sk-fake", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"free-model-alias": "free-model"}, + ) + assert _is_model_cost_zero(model="free-model", llm_router=router) is True + result = _is_model_cost_zero(model="free-model-alias", llm_router=router) + assert result is True, "Alias of an explicitly free model should bypass budget (return True)" + + def test_model_group_alias_to_paid_model_enforces_budget(self): + """An alias must not waive budget checks for a paid model group.""" + router = Router( + model_list=[ + { + "model_name": "paid-model", + "litellm_params": { + "model": "openai/gpt-4o-mini", + "api_key": "sk-fake", + }, + }, + ], + model_group_alias={"paid-model-alias": "paid-model"}, + ) + result = _is_model_cost_zero(model="paid-model-alias", llm_router=router) + assert result is False, "Alias of a paid model should enforce budget" + + def test_hidden_model_group_alias_enforces_budget(self): + """A hidden alias has no resolvable model group, so budget stays enforced.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "openai/nonexistent-but-free-model", + "api_key": "sk-fake", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"hidden-alias": {"model": "free-model", "hidden": True}}, + ) + result = _is_model_cost_zero(model="hidden-alias", llm_router=router) + assert result is False, "Hidden alias should enforce budget" + + def test_dangling_model_group_alias_enforces_budget(self): + """An alias pointing at a model group that does not exist must not bypass budget.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "openai/nonexistent-but-free-model", + "api_key": "sk-fake", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"dangling-alias": "no-such-model-group"}, + ) + result = _is_model_cost_zero(model="dangling-alias", llm_router=router) + assert result is False, "Alias to a missing model group should enforce budget" + + def test_model_group_alias_to_ptu_flat_cost_enforces_budget(self): + """A PTU deployment bills reserved capacity as a flat cost and carries an explicit + zero per-token price. Reaching it through an alias must still enforce budget.""" + router = Router( + model_list=[ + { + "model_name": "ptu-model", + "litellm_params": { + "model": "azure/gpt-4o", + "api_key": "sk-fake", + "api_base": "https://example.openai.azure.com", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": { + "id": "ptu-model-id", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + "ptu_count": 100, + "cost_per_ptu_per_hour": 2.0, + }, + }, + ], + model_group_alias={"ptu-model-alias": "ptu-model"}, + ) + assert _is_model_cost_zero(model="ptu-model", llm_router=router) is False + result = _is_model_cost_zero(model="ptu-model-alias", llm_router=router) + assert result is False, "Alias of a PTU flat-cost model should enforce budget" diff --git a/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py b/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py index 71671966d1a..e49fa462b53 100644 --- a/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py +++ b/tests/test_litellm/proxy/hooks/test_max_budget_limiter.py @@ -27,6 +27,7 @@ def _make_user_api_key_auth( user_spend: float = 0.0, team_id=None, budget_reservation=None, + skip_budget_checks: bool = False, ) -> UserAPIKeyAuth: return UserAPIKeyAuth( api_key="sk-test", @@ -35,6 +36,7 @@ def _make_user_api_key_auth( user_spend=user_spend, team_id=team_id, budget_reservation=budget_reservation, + skip_budget_checks=skip_budget_checks, ) @@ -235,3 +237,43 @@ async def test_no_max_budget_passes(): assert result is None mock_get_spend.assert_not_awaited() + + +# `get_current_spend` falls back to the caller-supplied `fallback_spend`, which the hook +# passes from `user_api_key_dict.user_spend`, so an over-budget request can be set up +# through the auth object alone -- no need to patch the SDK's own spend lookup. The pair +# below differs only in `skip_budget_checks`, which is what pins the exemption. + + +@pytest.mark.asyncio +async def test_budget_exempt_request_passes_when_over_budget(): + """A request auth marked exempt (zero-cost model) is admitted despite being over budget.""" + handler = _PROXY_MaxBudgetLimiter() + user_api_key_dict = _make_user_api_key_auth(user_max_budget=10.0, user_spend=99.0, skip_budget_checks=True) + + result = await handler.async_pre_call_hook( + user_api_key_dict=user_api_key_dict, + cache=DualCache(), + data={"model": "free-model"}, + call_type="completion", + ) + + assert result is None + + +@pytest.mark.asyncio +async def test_non_exempt_request_still_blocked_when_over_budget(): + """The same request without the exemption is still refused, so the flag is what admits it.""" + handler = _PROXY_MaxBudgetLimiter() + user_api_key_dict = _make_user_api_key_auth(user_max_budget=10.0, user_spend=99.0) + + with pytest.raises(HTTPException) as exc_info: + await handler.async_pre_call_hook( + user_api_key_dict=user_api_key_dict, + cache=DualCache(), + data={"model": "paid-model"}, + call_type="completion", + ) + + assert exc_info.value.status_code == 429 + assert "Max budget limit reached." in exc_info.value.detail diff --git a/tests/test_litellm/test_register_model_zero_cost_persistence.py b/tests/test_litellm/test_register_model_zero_cost_persistence.py index 15c8721a7f1..a5ad3a1f5e5 100644 --- a/tests/test_litellm/test_register_model_zero_cost_persistence.py +++ b/tests/test_litellm/test_register_model_zero_cost_persistence.py @@ -13,7 +13,7 @@ already-present sparse entry (e.g. router model id with only ``{"id": ..., "db_model": True}``), the synthesized zeros get written back, and the entry flips from "no cost keys" → "cost keys = 0". -That defeats ``_is_cost_explicitly_configured`` (added in #24949), which +That defeats ``_group_declares_explicit_cost`` (added in #24949), which checks whether the cost keys are present in the raw entry — after the write-back they are. ``_is_model_cost_zero`` then returns ``True`` and ``common_checks`` skips every tag / key / team / user / org budget check