fix(proxy): resolve model_group_alias in the zero-cost budget predicate (#43512)

`_is_model_cost_zero()` reads a group's cost through `Router.get_model_group_info()`,
which resolves `model_group_alias`, and then gates that on `_is_cost_explicitly_configured()`,
which scanned `Router.model_list` for an exact `model_name` match. Alias names live only in
`Router.model_group_alias` and are never `model_name` entries, so the scan found nothing and
returned False. That False means "the zero cost was defaulted, not configured" (the sparse
auto-registration gate added for #24770), so a model priced explicitly at 0 had budget
enforced against it when requested through an alias, while the same deployment under its own
name was exempt. Both names route to the same deployment and add nothing to spend.

The two lookups in one function disagreeing is the bug, so they now share one resolution:
`_is_cost_explicitly_configured()` resolves through `Router.get_model_list()`, the same
alias-aware path `get_model_group_info()` takes. That also reaches a deployment which prices
itself through its `model_info` block, whose cost-map entry lands under the deployment id.
`_group_declares_explicit_cost()` was an alias-aware copy of this function, wired only into
`model_has_no_cost_mapping()` and never into the budget path; its body is what
`_is_cost_explicitly_configured()` now carries, and both callers share it so the two cannot
drift apart again.

`_has_ptu_flat_cost()` scanned `model_list` the same way and runs after the gate above, so
resolving one without the other would let an aliased PTU group — explicit zero per-token
price alongside a flat capacity cost — pass as free. It resolves the same way now.

Tests cover the predicate and the request path it feeds: over-budget requests through
`_should_skip_budget_checks()` into `common_checks()` for an aliased free model (allowed) and
an aliased paid model (refused), the predicate for free, paid, PTU, hidden and dangling
aliases, and `model_has_no_cost_mapping()` through an alias so the other caller of the shared
check stays covered.

Unchanged: priced groups (the predicate returns False before the gate), unmapped groups whose
zero cost was defaulted (#24770), hidden aliases and aliases pointing at a nonexistent group
(`get_model_group_info()` returns None for both, so the cost is unknown and budget is
enforced), and non-aliased PTU groups.

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
fedaeho 2026-09-29 13:46:33 +09:00 • committed by GitHub
parent 7b2cbf6e7f
commit 85dc7cb62e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 257 additions and 27 deletions

View file

@ -567,10 +567,12 @@ def _has_ptu_flat_cost(model: str, llm_router: "Router") -> bool:
Such a deployment carries an explicit zero per-token price so the flat cost is not charged
twice, which otherwise reads here as a free model and waives every budget check for it.
Resolved through ``Router.get_model_list()``, which includes ``model_group_alias``, because
this runs after the explicit-cost gate: resolving that gate alone would let an aliased PTU
group through as free.
"""
for deployment in llm_router.model_list:
if deployment.get("model_name") != model:
continue
for deployment in llm_router.get_model_list(model_name=model) or ():
model_info = deployment.get("model_info") or _NO_MODEL_INFO
if model_info.get("ptu_count") is not None and model_info.get("cost_per_ptu_per_hour") is not None:
return True
@ -586,14 +588,19 @@ def _is_cost_explicitly_configured(model: str, llm_router: "Router") -> bool:
cost map, it creates a sparse entry like {"id": "<hash>"} with no cost
fields. _get_model_info_helper() then defaults missing costs to 0.
This function detects that scenario by checking the raw model_cost entry.
The group is resolved through ``Router.get_model_list()``, the same resolution
``get_model_group_info()`` applies when the caller reads the cost a few lines earlier, so the
two lookups cannot disagree: names defined in ``Router.model_group_alias`` are not
``model_name`` entries in ``Router.model_list``, and scanning that list by exact name reported
every aliased group as unconfigured. It also reaches a deployment that prices itself through
its ``model_info`` block, whose entry lands in the cost map under the deployment id.
"""
for deployment in llm_router.model_list:
if deployment.get("model_name") != model:
continue
model_id = deployment.get("model_info", {}).get("id")
for deployment in llm_router.get_model_list(model_name=model) or ():
model_id = (deployment.get("model_info") or _EMPTY_COST_ENTRY).get("id")
if model_id is None:
continue
raw_entry = litellm.model_cost.get(model_id, {})
raw_entry = litellm.model_cost.get(model_id, _EMPTY_COST_ENTRY)
if "input_cost_per_token" in raw_entry or "output_cost_per_token" in raw_entry:
return True
return False
@ -648,24 +655,6 @@ def _model_group_has_pricing(model: str, llm_router: "Router") -> bool:
return False
def _group_declares_explicit_cost(model: str, llm_router: "Router") -> bool:
"""
Alias-aware counterpart to ``_is_cost_explicitly_configured``, which resolves the model group
the same way ``_model_group_has_pricing`` does. A deployment that prices itself through its
``model_info`` block lands in the cost map under its deployment id rather than in its
litellm_params, and reaching that entry through the router's own resolution keeps an alias
pointing at such a group from being read as unpriced.
"""
for deployment in llm_router.get_model_list(model_name=model) or ():
model_id = (deployment.get("model_info") or _EMPTY_COST_ENTRY).get("id")
if model_id is None:
continue
raw_entry = litellm.model_cost.get(model_id, _EMPTY_COST_ENTRY)
if "input_cost_per_token" in raw_entry or "output_cost_per_token" in raw_entry:
return True
return False
def model_has_no_cost_mapping(model: str | None, llm_router: Router | None) -> bool:
if not model or llm_router is None:
return False
@ -676,7 +665,7 @@ def model_has_no_cost_mapping(model: str | None, llm_router: Router | None) -> b
if _model_group_has_pricing(model=model, llm_router=llm_router):
return False
return not _group_declares_explicit_cost(model=model, llm_router=llm_router)
return not _is_cost_explicitly_configured(model=model, llm_router=llm_router)
def _unpriced_models_in_request(model: str | list[str] | None, llm_router: Router | None) -> tuple[str, ...]:

View file

@ -8620,6 +8620,30 @@ def test_model_has_no_cost_mapping_unpriced_model_is_true():
assert model_has_no_cost_mapping(model="unpriced-group", llm_router=router) is True
def test_model_has_no_cost_mapping_resolves_model_group_alias():
"""This helper and the zero-cost budget predicate share one explicit-cost check, so the
alias resolution it depends on has to keep working for both."""
from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping
from litellm.router import Router
router = Router(
model_list=[
{
"model_name": "priced-group",
"litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-test"},
},
{
"model_name": "unpriced-group",
"litellm_params": {"model": UNPRICED_UNDERLYING_MODEL, "api_key": "sk-test"},
},
],
model_group_alias={"priced-alias": "priced-group", "unpriced-alias": "unpriced-group"},
)
assert model_has_no_cost_mapping(model="priced-alias", llm_router=router) is False
assert model_has_no_cost_mapping(model="unpriced-alias", llm_router=router) is True
def test_model_has_no_cost_mapping_no_model_or_router_is_false():
from litellm.proxy.auth.auth_checks import model_has_no_cost_mapping

View file

@ -190,6 +190,142 @@ class TestUnmappedModelBudgetEnforcement:
assert "input_cost_per_token" not in litellm.model_cost.get("alias-id", {})
assert _is_model_cost_zero(model="smart-router", llm_router=router) is False
def test_model_group_alias_to_free_model_bypasses_budget(self):
"""A zero-cost group reached through model_group_alias bypasses budget, like its own name.
Both names route to the same deployment and add nothing to spend, so refusing one of
them denies a request on spend it cannot produce.
"""
router = Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "ollama/llama2",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {"id": "free-model-id"},
},
],
model_group_alias={"free-model-alias": "free-model"},
)
assert _is_model_cost_zero(model="free-model", llm_router=router) is True
assert _is_model_cost_zero(model="free-model-alias", llm_router=router) is True, (
"An alias pointing at an explicitly-zero-cost group must be read as free, like its own name"
)
def test_model_group_alias_item_form_bypasses_budget(self):
"""The dict alias form ({"model": ..., "hidden": False}) resolves like the string form."""
router = Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "ollama/llama2",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {"id": "free-model-id"},
},
],
model_group_alias={"free-model-alias": {"model": "free-model", "hidden": False}},
)
assert _is_model_cost_zero(model="free-model-alias", llm_router=router) is True
def test_model_group_alias_to_paid_model_enforces_budget(self):
"""An alias does not turn a priced group into a free one."""
router = Router(
model_list=[
{
"model_name": "paid-model",
"litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-fake"},
"model_info": {"id": "paid-model-id"},
},
],
model_group_alias={"paid-model-alias": "paid-model"},
)
assert _is_model_cost_zero(model="paid-model-alias", llm_router=router) is False
def test_model_group_alias_to_ptu_flat_cost_enforces_budget(self):
"""A PTU group keeps budget enforced through an alias.
Its explicit zero per-token price exists so the flat capacity cost is not charged twice,
so the PTU check has to resolve the alias too — resolving only the explicit-cost gate
would let this through as free.
"""
router = Router(
model_list=[
{
"model_name": "ptu-model",
"litellm_params": {
"model": "azure/ptu-deployment",
"api_base": "https://fake.openai.azure.com",
"api_key": "sk-fake",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {
"id": "ptu-model-id",
"ptu_count": 100,
"cost_per_ptu_per_hour": 2.0,
},
},
],
model_group_alias={"ptu-model-alias": "ptu-model"},
)
assert _is_model_cost_zero(model="ptu-model", llm_router=router) is False
assert _is_model_cost_zero(model="ptu-model-alias", llm_router=router) is False, (
"An aliased PTU group must not be read as free"
)
def test_hidden_model_group_alias_enforces_budget(self):
"""A hidden alias keeps budget enforced: get_model_group_info() returns None for it,
so the cost is unknown before the configuration gate is reached."""
router = Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "ollama/llama2",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {"id": "free-model-id"},
},
],
model_group_alias={"hidden-alias": {"model": "free-model", "hidden": True}},
)
assert _is_model_cost_zero(model="hidden-alias", llm_router=router) is False
def test_dangling_model_group_alias_enforces_budget(self):
"""An alias pointing at a group that does not exist keeps budget enforced."""
router = Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "ollama/llama2",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {"id": "free-model-id"},
},
],
model_group_alias={"dangling-alias": "model-that-does-not-exist"},
)
assert _is_model_cost_zero(model="dangling-alias", llm_router=router) is False
def test_handles_router_without_zero_cost_cache_attribute(self):
"""Tolerate router-like objects (e.g. ``MagicMock`` stand-ins) that
do not expose ``_zero_cost_cache`` — the auth check must still

View file

@ -588,3 +588,84 @@ class TestEdgeCases:
request=MagicMock(),
)
assert result is True
class TestOverBudgetRequestThroughModelGroupAlias:
"""The whole path a request takes, not just the predicate.
`user_api_key_auth._should_skip_budget_checks()` derives the exemption from the requested
model name and `common_checks()` enforces the budgets with it, so a break anywhere between
alias resolution and enforcement shows up here. See
https://github.com/BerriAI/litellm/issues/35369.
"""
ROUTE = "/v1/chat/completions"
@staticmethod
def _router() -> Router:
return Router(
model_list=[
{
"model_name": "free-model",
"litellm_params": {
"model": "ollama/llama2",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0.0,
"output_cost_per_token": 0.0,
},
"model_info": {"id": "free-model-id"},
},
{
"model_name": "paid-model",
"litellm_params": {"model": "gpt-3.5-turbo", "api_key": "sk-test"},
"model_info": {"id": "paid-model-id"},
},
],
model_group_alias={"free-model-alias": "free-model", "paid-model-alias": "paid-model"},
)
async def _request(self, model: str, proxy_logging) -> bool:
"""Run one over-budget request for `model`, deriving the exemption the way auth does."""
from litellm.proxy.auth.user_api_key_auth import _should_skip_budget_checks
router = self._router()
request_data = {"model": model}
skip_budget_checks = _should_skip_budget_checks(
request_data=request_data, route=self.ROUTE, request=None, llm_router=router
)
return await common_checks(
request_body=request_data,
team_object=None,
user_object=LiteLLM_UserTable(user_id="test-user", spend=100.0, max_budget=50.0),
end_user_object=None,
global_proxy_spend=None,
general_settings={},
route=self.ROUTE,
llm_router=router,
proxy_logging_obj=proxy_logging,
valid_token=UserAPIKeyAuth(token="test-token", user_id="test-user"),
request=MagicMock(),
skip_budget_checks=skip_budget_checks,
)
@pytest.mark.asyncio
async def test_over_budget_request_for_aliased_free_model_is_allowed(self, mock_proxy_logging):
assert await self._request("free-model-alias", mock_proxy_logging) is True
@pytest.mark.asyncio
async def test_over_budget_request_for_free_model_is_allowed(self, mock_proxy_logging):
"""The same deployment under its own name, so the alias is the only difference above."""
assert await self._request("free-model", mock_proxy_logging) is True
@pytest.mark.asyncio
async def test_over_budget_request_for_aliased_paid_model_is_blocked(self, mock_proxy_logging):
with pytest.raises(litellm.BudgetExceededError) as exc_info:
await self._request("paid-model-alias", mock_proxy_logging)
assert exc_info.value.current_cost == 100.0
assert exc_info.value.max_budget == 50.0
@pytest.mark.asyncio
async def test_over_budget_request_for_paid_model_is_blocked(self, mock_proxy_logging):
with pytest.raises(litellm.BudgetExceededError):
await self._request("paid-model", mock_proxy_logging)