From 85dc7cb62efa52794982af90164eb19a41140b80 Mon Sep 17 00:00:00 2001 From: fedaeho <39611158+fedaeho@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:46:33 +0900 Subject: [PATCH] fix(proxy): resolve model_group_alias in the zero-cost budget predicate (#43512) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_is_model_cost_zero()` reads a group's cost through `Router.get_model_group_info()`, which resolves `model_group_alias`, and then gates that on `_is_cost_explicitly_configured()`, which scanned `Router.model_list` for an exact `model_name` match. Alias names live only in `Router.model_group_alias` and are never `model_name` entries, so the scan found nothing and returned False. That False means "the zero cost was defaulted, not configured" (the sparse auto-registration gate added for #24770), so a model priced explicitly at 0 had budget enforced against it when requested through an alias, while the same deployment under its own name was exempt. Both names route to the same deployment and add nothing to spend. The two lookups in one function disagreeing is the bug, so they now share one resolution: `_is_cost_explicitly_configured()` resolves through `Router.get_model_list()`, the same alias-aware path `get_model_group_info()` takes. That also reaches a deployment which prices itself through its `model_info` block, whose cost-map entry lands under the deployment id. `_group_declares_explicit_cost()` was an alias-aware copy of this function, wired only into `model_has_no_cost_mapping()` and never into the budget path; its body is what `_is_cost_explicitly_configured()` now carries, and both callers share it so the two cannot drift apart again. `_has_ptu_flat_cost()` scanned `model_list` the same way and runs after the gate above, so resolving one without the other would let an aliased PTU group — explicit zero per-token price alongside a flat capacity cost — pass as free. It resolves the same way now. Tests cover the predicate and the request path it feeds: over-budget requests through `_should_skip_budget_checks()` into `common_checks()` for an aliased free model (allowed) and an aliased paid model (refused), the predicate for free, paid, PTU, hidden and dangling aliases, and `model_has_no_cost_mapping()` through an alias so the other caller of the shared check stays covered. Unchanged: priced groups (the predicate returns False before the gate), unmapped groups whose zero cost was defaulted (#24770), hidden aliases and aliases pointing at a nonexistent group (`get_model_group_info()` returns None for both, so the cost is unknown and budget is enforced), and non-aliased PTU groups. Co-authored-by: Claude Opus 5 (1M context) --- litellm/proxy/auth/auth_checks.py | 43 +++--- .../proxy/auth/test_auth_checks.py | 24 ++++ .../test_unmapped_model_budget_enforcement.py | 136 ++++++++++++++++++ .../test_zero_cost_model_budget_bypass.py | 81 +++++++++++ 4 files changed, 257 insertions(+), 27 deletions(-) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 12d420141f1..51cc70c010b 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -567,10 +567,12 @@ def _has_ptu_flat_cost(model: str, llm_router: "Router") -> bool: Such a deployment carries an explicit zero per-token price so the flat cost is not charged twice, which otherwise reads here as a free model and waives every budget check for it. + + Resolved through ``Router.get_model_list()``, which includes ``model_group_alias``, because + this runs after the explicit-cost gate: resolving that gate alone would let an aliased PTU + group through as free. """ - for deployment in llm_router.model_list: - if deployment.get("model_name") != model: - continue + for deployment in llm_router.get_model_list(model_name=model) or (): model_info = deployment.get("model_info") or _NO_MODEL_INFO if model_info.get("ptu_count") is not None and model_info.get("cost_per_ptu_per_hour") is not None: return True @@ -586,14 +588,19 @@ def _is_cost_explicitly_configured(model: str, llm_router: "Router") -> bool: cost map, it creates a sparse entry like {"id": ""} with no cost fields. _get_model_info_helper() then defaults missing costs to 0. This function detects that scenario by checking the raw model_cost entry. + + The group is resolved through ``Router.get_model_list()``, the same resolution + ``get_model_group_info()`` applies when the caller reads the cost a few lines earlier, so the + two lookups cannot disagree: names defined in ``Router.model_group_alias`` are not + ``model_name`` entries in ``Router.model_list``, and scanning that list by exact name reported + every aliased group as unconfigured. It also reaches a deployment that prices itself through + its ``model_info`` block, whose entry lands in the cost map under the deployment id. """ - for deployment in llm_router.model_list: - if deployment.get("model_name") != model: - continue - model_id = deployment.get("model_info", {}).get("id") + for deployment in llm_router.get_model_list(model_name=model) or (): + model_id = (deployment.get("model_info") or _EMPTY_COST_ENTRY).get("id") if model_id is None: continue - raw_entry = litellm.model_cost.get(model_id, {}) + raw_entry = litellm.model_cost.get(model_id, _EMPTY_COST_ENTRY) if "input_cost_per_token" in raw_entry or "output_cost_per_token" in raw_entry: return True return False @@ -648,24 +655,6 @@ def _model_group_has_pricing(model: str, llm_router: "Router") -> bool: return False -def _group_declares_explicit_cost(model: str, llm_router: "Router") -> bool: - """ - Alias-aware counterpart to ``_is_cost_explicitly_configured``, which resolves the model group - the same way ``_model_group_has_pricing`` does. A deployment that prices itself through its - ``model_info`` block lands in the cost map under its deployment id rather than in its - litellm_params, and reaching that entry through the router's own resolution keeps an alias - pointing at such a group from being read as unpriced. - """ - for deployment in llm_router.get_model_list(model_name=model) or (): - model_id = (deployment.get("model_info") or _EMPTY_COST_ENTRY).get("id") - if model_id is None: - continue - raw_entry = litellm.model_cost.get(model_id, _EMPTY_COST_ENTRY) - if "input_cost_per_token" in raw_entry or "output_cost_per_token" in raw_entry: - return True - return False - - def model_has_no_cost_mapping(model: str | None, llm_router: Router | None) -> bool: if not model or llm_router is None: return False @@ -676,7 +665,7 @@ def model_has_no_cost_mapping(model: str | None, llm_router: Router | None) -> b if _model_group_has_pricing(model=model, llm_router=llm_router): return False - return not _group_declares_explicit_cost(model=model, llm_router=llm_router) + return not _is_cost_explicitly_configured(model=model, llm_router=llm_router) def _unpriced_models_in_request(model: str | list[str] | None, llm_router: Router | None) -> tuple[str, ...]: diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index f014e9c26d1..dabd97cff0b 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -8620,6 +8620,30 @@ def test_model_has_no_cost_mapping_unpriced_model_is_true(): assert model_has_no_cost_mapping(model="unpriced-group", llm_router=router) is True +def test_model_has_no_cost_mapping_resolves_model_group_alias(): + """This helper and the zero-cost budget predicate share one explicit-cost check, so the + alias resolution it depends on has to keep working for both.""" + from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping + from litellm.router import Router + + router = Router( + model_list=[ + { + "model_name": "priced-group", + "litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-test"}, + }, + { + "model_name": "unpriced-group", + "litellm_params": {"model": UNPRICED_UNDERLYING_MODEL, "api_key": "sk-test"}, + }, + ], + model_group_alias={"priced-alias": "priced-group", "unpriced-alias": "unpriced-group"}, + ) + + assert model_has_no_cost_mapping(model="priced-alias", llm_router=router) is False + assert model_has_no_cost_mapping(model="unpriced-alias", llm_router=router) is True + + def test_model_has_no_cost_mapping_no_model_or_router_is_false(): from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping diff --git a/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py b/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py index bbe343bcede..7665008a6a6 100644 --- a/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py +++ b/tests/test_litellm/proxy/auth/test_unmapped_model_budget_enforcement.py @@ -190,6 +190,142 @@ class TestUnmappedModelBudgetEnforcement: assert "input_cost_per_token" not in litellm.model_cost.get("alias-id", {}) assert _is_model_cost_zero(model="smart-router", llm_router=router) is False + def test_model_group_alias_to_free_model_bypasses_budget(self): + """A zero-cost group reached through model_group_alias bypasses budget, like its own name. + + Both names route to the same deployment and add nothing to spend, so refusing one of + them denies a request on spend it cannot produce. + """ + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "ollama/llama2", + "api_base": "http://localhost:11434", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"free-model-alias": "free-model"}, + ) + + assert _is_model_cost_zero(model="free-model", llm_router=router) is True + assert _is_model_cost_zero(model="free-model-alias", llm_router=router) is True, ( + "An alias pointing at an explicitly-zero-cost group must be read as free, like its own name" + ) + + def test_model_group_alias_item_form_bypasses_budget(self): + """The dict alias form ({"model": ..., "hidden": False}) resolves like the string form.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "ollama/llama2", + "api_base": "http://localhost:11434", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"free-model-alias": {"model": "free-model", "hidden": False}}, + ) + + assert _is_model_cost_zero(model="free-model-alias", llm_router=router) is True + + def test_model_group_alias_to_paid_model_enforces_budget(self): + """An alias does not turn a priced group into a free one.""" + router = Router( + model_list=[ + { + "model_name": "paid-model", + "litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-fake"}, + "model_info": {"id": "paid-model-id"}, + }, + ], + model_group_alias={"paid-model-alias": "paid-model"}, + ) + + assert _is_model_cost_zero(model="paid-model-alias", llm_router=router) is False + + def test_model_group_alias_to_ptu_flat_cost_enforces_budget(self): + """A PTU group keeps budget enforced through an alias. + + Its explicit zero per-token price exists so the flat capacity cost is not charged twice, + so the PTU check has to resolve the alias too — resolving only the explicit-cost gate + would let this through as free. + """ + router = Router( + model_list=[ + { + "model_name": "ptu-model", + "litellm_params": { + "model": "azure/ptu-deployment", + "api_base": "https://fake.openai.azure.com", + "api_key": "sk-fake", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": { + "id": "ptu-model-id", + "ptu_count": 100, + "cost_per_ptu_per_hour": 2.0, + }, + }, + ], + model_group_alias={"ptu-model-alias": "ptu-model"}, + ) + + assert _is_model_cost_zero(model="ptu-model", llm_router=router) is False + assert _is_model_cost_zero(model="ptu-model-alias", llm_router=router) is False, ( + "An aliased PTU group must not be read as free" + ) + + def test_hidden_model_group_alias_enforces_budget(self): + """A hidden alias keeps budget enforced: get_model_group_info() returns None for it, + so the cost is unknown before the configuration gate is reached.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "ollama/llama2", + "api_base": "http://localhost:11434", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"hidden-alias": {"model": "free-model", "hidden": True}}, + ) + + assert _is_model_cost_zero(model="hidden-alias", llm_router=router) is False + + def test_dangling_model_group_alias_enforces_budget(self): + """An alias pointing at a group that does not exist keeps budget enforced.""" + router = Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "ollama/llama2", + "api_base": "http://localhost:11434", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + ], + model_group_alias={"dangling-alias": "model-that-does-not-exist"}, + ) + + assert _is_model_cost_zero(model="dangling-alias", llm_router=router) is False + def test_handles_router_without_zero_cost_cache_attribute(self): """Tolerate router-like objects (e.g. ``MagicMock`` stand-ins) that do not expose ``_zero_cost_cache`` — the auth check must still diff --git a/tests/unit/proxy/test_zero_cost_model_budget_bypass.py b/tests/unit/proxy/test_zero_cost_model_budget_bypass.py index 51a7cb2ee9d..56133f2d35b 100644 --- a/tests/unit/proxy/test_zero_cost_model_budget_bypass.py +++ b/tests/unit/proxy/test_zero_cost_model_budget_bypass.py @@ -588,3 +588,84 @@ class TestEdgeCases: request=MagicMock(), ) assert result is True + + +class TestOverBudgetRequestThroughModelGroupAlias: + """The whole path a request takes, not just the predicate. + + `user_api_key_auth._should_skip_budget_checks()` derives the exemption from the requested + model name and `common_checks()` enforces the budgets with it, so a break anywhere between + alias resolution and enforcement shows up here. See + https://github.com/BerriAI/litellm/issues/35369. + """ + + ROUTE = "/v1/chat/completions" + + @staticmethod + def _router() -> Router: + return Router( + model_list=[ + { + "model_name": "free-model", + "litellm_params": { + "model": "ollama/llama2", + "api_base": "http://localhost:11434", + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + }, + "model_info": {"id": "free-model-id"}, + }, + { + "model_name": "paid-model", + "litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-test"}, + "model_info": {"id": "paid-model-id"}, + }, + ], + model_group_alias={"free-model-alias": "free-model", "paid-model-alias": "paid-model"}, + ) + + async def _request(self, model: str, proxy_logging) -> bool: + """Run one over-budget request for `model`, deriving the exemption the way auth does.""" + from litellm.proxy.auth.user_api_key_auth import _should_skip_budget_checks + + router = self._router() + request_data = {"model": model} + skip_budget_checks = _should_skip_budget_checks( + request_data=request_data, route=self.ROUTE, request=None, llm_router=router + ) + return await common_checks( + request_body=request_data, + team_object=None, + user_object=LiteLLM_UserTable(user_id="test-user", spend=100.0, max_budget=50.0), + end_user_object=None, + global_proxy_spend=None, + general_settings={}, + route=self.ROUTE, + llm_router=router, + proxy_logging_obj=proxy_logging, + valid_token=UserAPIKeyAuth(token="test-token", user_id="test-user"), + request=MagicMock(), + skip_budget_checks=skip_budget_checks, + ) + + @pytest.mark.asyncio + async def test_over_budget_request_for_aliased_free_model_is_allowed(self, mock_proxy_logging): + assert await self._request("free-model-alias", mock_proxy_logging) is True + + @pytest.mark.asyncio + async def test_over_budget_request_for_free_model_is_allowed(self, mock_proxy_logging): + """The same deployment under its own name, so the alias is the only difference above.""" + assert await self._request("free-model", mock_proxy_logging) is True + + @pytest.mark.asyncio + async def test_over_budget_request_for_aliased_paid_model_is_blocked(self, mock_proxy_logging): + with pytest.raises(litellm.BudgetExceededError) as exc_info: + await self._request("paid-model-alias", mock_proxy_logging) + + assert exc_info.value.current_cost == 100.0 + assert exc_info.value.max_budget == 50.0 + + @pytest.mark.asyncio + async def test_over_budget_request_for_paid_model_is_blocked(self, mock_proxy_logging): + with pytest.raises(litellm.BudgetExceededError): + await self._request("paid-model", mock_proxy_logging)