mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(router): never price a strategy-router alias (#36691)
* fix: never price a strategy-router alias A strategy-router alias (auto_router/complexity_router/<name>) is never the deployment that gets called or billed, but custom pricing configured on it was being treated as real pricing in two places: - registered in litellm.model_cost under the alias deployment id, so an explicit zero made _is_cost_explicitly_configured() report the group as a genuinely free model and every budget check was skipped, while the request routed to a paid deployment and accrued real spend - copied onto request_kwargs by the alias-params merge, so the routed deployment got re-registered at the alias price and the request billed 0.0 Both are fixed at the writer, so config, /model/new and price-map reload all take the same path Co-Authored-By: Claude <noreply@anthropic.com> * chore: annotate filtered cost-map copy for the mutable-collection gate Co-Authored-By: Claude <noreply@anthropic.com> --------- Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
parent
c642e8400d
commit
1911269ddf
4 changed files with 140 additions and 2 deletions
|
|
@ -8612,7 +8612,18 @@ class Router:
|
|||
Nothing is recorded for replay: a refresh walks the live routers instead,
|
||||
so a deleted, repointed or never-added deployment, and a discarded router,
|
||||
drop out of the rebuild on their own.
|
||||
|
||||
A strategy-router alias is never the deployment actually called or
|
||||
billed, so custom pricing configured on it must not become a cost-map
|
||||
price: an explicit zero would let ``_is_cost_explicitly_configured``
|
||||
treat the alias as a genuinely free model and waive budget checks for
|
||||
requests that route to (and bill as) a real deployment.
|
||||
"""
|
||||
if classify_strategy_router_model(model) is not None:
|
||||
model_info = { # mutable-ok: filtered copy of the caller's entry, handed straight to register_model
|
||||
k: v for k, v in model_info.items() if k not in CustomPricingLiteLLMParams.model_fields
|
||||
}
|
||||
|
||||
if model_id is not None:
|
||||
litellm.register_model(model_cost={model_id: model_info}, persist_across_reloads=False)
|
||||
|
||||
|
|
@ -11439,13 +11450,16 @@ class Router:
|
|||
# deployment the hook selected won't have them. Router-only fields
|
||||
# (tpm, rpm, weight, complexity_router_config, ...) are excluded from the
|
||||
# actual outbound LLM call downstream by litellm.types.utils.all_litellm_params,
|
||||
# not here.
|
||||
# not here. Custom pricing fields ARE call params, so they must be
|
||||
# excluded here: they price the alias, not the deployment the hook
|
||||
# selected, and forwarding them re-registers the routed deployment at
|
||||
# the alias's price (an explicit 0 makes every alias request bill $0).
|
||||
if pre_routing_hook_response is not None:
|
||||
alias_index: Final = self.model_name_to_deployment_indices.get(model, [])
|
||||
if alias_index:
|
||||
alias_litellm_params: Final = self.model_list[alias_index[0]].get("litellm_params", {})
|
||||
for key, value in alias_litellm_params.items():
|
||||
if key != "model" and value is not None:
|
||||
if key != "model" and key not in CustomPricingLiteLLMParams.model_fields and value is not None:
|
||||
request_kwargs.setdefault(key, value)
|
||||
|
||||
return pre_routing_hook_response
|
||||
|
|
|
|||
|
|
@ -162,6 +162,34 @@ class TestUnmappedModelBudgetEnforcement:
|
|||
# Subsequent call sees the new pricing and enforces budget.
|
||||
assert _is_model_cost_zero(model="ramping-model", llm_router=router) is False
|
||||
|
||||
def test_strategy_router_alias_with_zero_pricing_enforces_budget(self):
|
||||
"""An auto-router alias is never the deployment that gets called or
|
||||
billed, so zero pricing configured on it must not waive budget checks
|
||||
for requests that route to (and bill as) a real paid deployment."""
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "smart-router",
|
||||
"litellm_params": {
|
||||
"model": "auto_router/complexity_router/smart-router",
|
||||
"complexity_router_default_model": "paid-model",
|
||||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"complexity_router_config": {"tiers": {"simple": "paid-model"}},
|
||||
},
|
||||
"model_info": {"id": "alias-id"},
|
||||
},
|
||||
{
|
||||
"model_name": "paid-model",
|
||||
"litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "paid-id"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
assert "input_cost_per_token" not in litellm.model_cost.get("alias-id", {})
|
||||
assert _is_model_cost_zero(model="smart-router", llm_router=router) is False
|
||||
|
||||
def test_handles_router_without_zero_cost_cache_attribute(self):
|
||||
"""Tolerate router-like objects (e.g. ``MagicMock`` stand-ins) that
|
||||
do not expose ``_zero_cost_cache`` — the auth check must still
|
||||
|
|
|
|||
|
|
@ -2040,6 +2040,44 @@ class TestRouterPreRoutingAliasOverrides:
|
|||
assert request_kwargs["drop_params"] is True
|
||||
assert request_kwargs["cache_control_injection_points"] == [{"location": "message", "role": "system"}]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_alias_custom_pricing_is_not_applied_to_request_kwargs(self):
|
||||
"""Custom pricing on the alias prices the alias, not the tier deployment
|
||||
the hook picked. Unlike the router-only fields, pricing fields are real
|
||||
call params, so forwarding them would re-register the routed deployment
|
||||
at the alias's price - an explicit 0 billing every request as free."""
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "smart-router",
|
||||
"litellm_params": {
|
||||
"model": "auto_router/complexity_router",
|
||||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"input_cost_per_second": 0.0,
|
||||
"drop_params": True,
|
||||
"complexity_router_config": {"tiers": {"SIMPLE": "gpt-4o-mini"}},
|
||||
"complexity_router_default_model": "gpt-4o",
|
||||
},
|
||||
},
|
||||
{"model_name": "gpt-4o-mini", "litellm_params": {"model": "openai/gpt-4o-mini"}},
|
||||
{"model_name": "gpt-4o", "litellm_params": {"model": "openai/gpt-4o"}},
|
||||
]
|
||||
)
|
||||
request_kwargs: dict = {}
|
||||
|
||||
result = await router.async_pre_routing_hook(
|
||||
model="smart-router",
|
||||
request_kwargs=request_kwargs,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
# Non-pricing alias params still carry over.
|
||||
assert request_kwargs["drop_params"] is True
|
||||
for field in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"):
|
||||
assert field not in request_kwargs
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_alias_overrides_exclude_only_model(self):
|
||||
"""`model` (the alias marker, e.g. auto_router/complexity_router) is
|
||||
|
|
|
|||
|
|
@ -1471,3 +1471,61 @@ def test_replay_live_router_model_cost_rebuilds_every_live_router():
|
|||
finally:
|
||||
litellm.model_cost = saved_model_cost
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_strategy_router_alias_pricing_never_enters_model_cost(monkeypatch):
|
||||
"""
|
||||
A strategy-router alias is never the deployment actually called or billed,
|
||||
so custom pricing configured on it must not be registered under its
|
||||
model_id - an explicit zero there makes the budget check treat the alias
|
||||
as a genuinely free model while requests bill as a real deployment. The
|
||||
strip must also survive a price-data reload, which rebuilds entries by
|
||||
walking the live routers.
|
||||
"""
|
||||
from litellm import utils as litellm_utils
|
||||
monkeypatch.setattr(
|
||||
litellm_utils,
|
||||
"_runtime_registered_model_cost",
|
||||
dict(litellm_utils._runtime_registered_model_cost),
|
||||
)
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "smart-router",
|
||||
"litellm_params": {
|
||||
"model": "auto_router/complexity_router/smart-router",
|
||||
"complexity_router_default_model": "paid-model",
|
||||
"input_cost_per_token": 0.0,
|
||||
"output_cost_per_token": 0.0,
|
||||
"complexity_router_config": {"tiers": {"simple": "paid-model"}},
|
||||
},
|
||||
"model_info": {"id": "strategy-alias-id", "max_input_tokens": 128000},
|
||||
},
|
||||
{
|
||||
"model_name": "paid-model",
|
||||
"litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "strategy-alias-paid-id"},
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
def _assert_alias_unpriced():
|
||||
entry = litellm.model_cost.get("strategy-alias-id")
|
||||
assert entry is not None, "Alias metadata should still be registered"
|
||||
assert entry["max_input_tokens"] == 128000
|
||||
assert "input_cost_per_token" not in entry
|
||||
assert "output_cost_per_token" not in entry
|
||||
|
||||
_assert_alias_unpriced()
|
||||
|
||||
saved_model_cost = litellm.model_cost
|
||||
try:
|
||||
_simulate_price_data_reload(
|
||||
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
||||
)
|
||||
_assert_alias_unpriced()
|
||||
assert router.model_list
|
||||
finally:
|
||||
litellm.model_cost = saved_model_cost
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue