mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(router): price a model group from the deployments that serve it (#44732)
* fix(router): price a model group from the deployments that serve it A model group's info read composed the group's own model_group_alias entry into its deployments, a hop the router never takes: an alias is resolved exactly once at request time, so a group reached as an alias target is served by its own deployments. For the chain X -> T -> U the price read for X included U's deployments too, and the free-model budget waiver refused a free request to X on an over-budget key, while GET /model_group/info reported U's providers and price for X. The group info read now prices a group from the deployments routing serves it with: the ones named after it, the routing group of that name, or the wildcard route matching it when neither exists. The budget waiver, GET /model_group/info, the rate limiters, and the response headers all read the same set as routing. get_model_list keeps its behavior for every other caller. * test(router): give the paid fixtures explicit per-token prices * test(router): call the routed-group read by name so the router coverage gate sees it * test(integration): audit the alias chain budget waiver on every route, shape, and outage Thirty-four cells under the management group prove an over-budget key is served through an alias chain entry at the price of the deployment that serves it, on chat, responses, and messages, sync and streamed, through the OpenAI and Anthropic SDKs and raw httpx on both replicas, with the chain middle, the reverse chain, a cost-map priced middle, a ghost middle, wildcard and routing-group targets, malformed and hostile inputs, a cached reply, a repointed alias, a provider failure, and two chaos bursts (a killed worker, a scripted outage) --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
c2c0bb583e
commit
c13746c951
5 changed files with 1204 additions and 18 deletions
|
|
@ -11276,9 +11276,7 @@ class Router:
|
|||
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None
|
||||
reasoning_efforts_initialized = False
|
||||
reasoning_efforts_unknown = False
|
||||
model_list: Final = self.get_model_list(model_name=model_group)
|
||||
if model_list is None:
|
||||
return None
|
||||
model_list: Final = self.get_model_list_of_routed_group(model_group)
|
||||
for model in model_list:
|
||||
is_match = False
|
||||
if (
|
||||
|
|
@ -12409,27 +12407,43 @@ class Router:
|
|||
returned_models.extend(self.get_model_list_from_model_alias(model_name=model_name))
|
||||
returned_models.extend(self.get_model_list_from_routing_groups(model_name=model_name))
|
||||
|
||||
if len(returned_models) == 0: # check if wildcard route
|
||||
potential_wildcard_models: Final = self.pattern_router.get_deployments_by_pattern(model=model_name or "")
|
||||
|
||||
## check for team-specific wildcard models
|
||||
if team_id is not None and team_id in self.team_pattern_routers:
|
||||
potential_team_only_wildcard_models: Final = self.team_pattern_routers[
|
||||
team_id
|
||||
].get_deployments_by_pattern(model=model_name or "")
|
||||
potential_wildcard_models.extend(potential_team_only_wildcard_models)
|
||||
|
||||
if model_name is not None and potential_wildcard_models is not None:
|
||||
for m in potential_wildcard_models:
|
||||
deployment_typed_dict = DeploymentTypedDict(**m)
|
||||
deployment_typed_dict["model_name"] = model_name
|
||||
returned_models.append(deployment_typed_dict)
|
||||
if len(returned_models) == 0 and model_name is not None:
|
||||
returned_models.extend(self._get_wildcard_deployments(model_name=model_name, team_id=team_id))
|
||||
|
||||
if model_name is None:
|
||||
returned_models += self.model_list
|
||||
|
||||
return returned_models
|
||||
|
||||
def _get_wildcard_deployments(self, model_name: str, team_id: str | None = None) -> list[DeploymentTypedDict]:
|
||||
"""
|
||||
The deployments of the wildcard routes matching model_name (the proxy-wide
|
||||
ones, plus team_id's own when given), each emitted under model_name.
|
||||
"""
|
||||
team_router: Final = self.team_pattern_routers.get(team_id) if team_id is not None else None
|
||||
matches: Final = [
|
||||
*self.pattern_router.get_deployments_by_pattern(model=model_name),
|
||||
*(team_router.get_deployments_by_pattern(model=model_name) if team_router is not None else ()),
|
||||
]
|
||||
return [{**DeploymentTypedDict(**m), "model_name": model_name} for m in matches]
|
||||
|
||||
def get_model_list_of_routed_group(self, model_group: str) -> list[DeploymentTypedDict]:
|
||||
"""
|
||||
The deployments a request the router has already resolved to model_group is
|
||||
served from: the ones named model_group, the routing group of that name, or
|
||||
the wildcard route matching it when neither exists.
|
||||
|
||||
Unlike get_model_list, model_group's own model_group_alias entry is not
|
||||
followed. The router resolves an alias exactly once, so a group reached as
|
||||
an alias target is served by its own deployments, never by a second hop:
|
||||
in the chain X -> T -> U a request to X is served from T's deployments.
|
||||
"""
|
||||
named: Final = [
|
||||
*self._get_all_deployments(model_name=model_group),
|
||||
*self.get_model_list_from_routing_groups(model_name=model_group),
|
||||
]
|
||||
return named or self._get_wildcard_deployments(model_name=model_group)
|
||||
|
||||
def resolved_litellm_models(self, model_name: str, team_id: str | None = None) -> tuple[str, ...]:
|
||||
"""The provider model strings `model_name` can actually be served by on this proxy.
|
||||
|
||||
|
|
|
|||
|
|
@ -105,6 +105,7 @@ ignored_function_names = [
|
|||
"_aanthropic_messages_retry_same_group", # Tested through the dropped-before-content retry tests in test_router.py
|
||||
"_aanthropic_messages_yield_recovered", # Tested through every mid-stream retry and fallback test in test_router.py
|
||||
"_anthropic_messages_policy_retries", # Tested through the retry budget precedence test in test_router.py
|
||||
"_get_wildcard_deployments", # Tested through the get_model_list_of_routed_group wildcard test in test_router.py
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
1018
tests/integration/authorization/test_alias_chain_budget_waiver.py
Normal file
1018
tests/integration/authorization/test_alias_chain_budget_waiver.py
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -32,6 +32,7 @@ from litellm.proxy._types import (
|
|||
from litellm.proxy.utils import PrismaClient
|
||||
from litellm.proxy.auth.auth_checks import (
|
||||
can_team_access_model,
|
||||
_is_model_cost_zero,
|
||||
_virtual_key_soft_budget_check,
|
||||
_team_soft_budget_check,
|
||||
)
|
||||
|
|
@ -1595,3 +1596,44 @@ async def test_get_user_object_cache_miss_emits_exactly_one_postgres_get_user_ob
|
|||
assert result is not None and result.user_id == user_id
|
||||
assert await _db_service_call_types(db_success_hook) == ("get_user_object",)
|
||||
assert db_success_hook.await_args_list[0].kwargs["parent_otel_span"] == "auth-span"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("entry_first", [True, False])
|
||||
def test_is_model_cost_zero_judges_an_alias_chain_by_the_deployment_its_entry_routes_to(
|
||||
monkeypatch: pytest.MonkeyPatch, entry_first: bool
|
||||
) -> None:
|
||||
"""chain-entry resolves one hop to local-free and is served by local-free's own free
|
||||
deployment, so an over-budget key is waived for it; local-free by name resolves to paid-gpt
|
||||
and stays enforced. local-free's alias is a hop the router never takes for chain-entry, and
|
||||
the per-name verdict cache must not let either name's verdict leak into the other's."""
|
||||
monkeypatch.setattr(litellm, "model_cost", dict(litellm.model_cost))
|
||||
router: Final = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "local-free",
|
||||
"litellm_params": {
|
||||
"model": "ollama/qwen3:0.6b",
|
||||
"api_base": "http://localhost:11434",
|
||||
"input_cost_per_token": 0,
|
||||
"output_cost_per_token": 0,
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "paid-gpt",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_key": "fake",
|
||||
"input_cost_per_token": 3e-06,
|
||||
"output_cost_per_token": 1.5e-05,
|
||||
},
|
||||
},
|
||||
],
|
||||
model_group_alias={"chain-entry": "local-free", "local-free": "paid-gpt"},
|
||||
)
|
||||
expected: Final = {"chain-entry": True, "local-free": False, "paid-gpt": False}
|
||||
order: Final = ("chain-entry", "local-free", "paid-gpt") if entry_first else ("local-free", "paid-gpt", "chain-entry")
|
||||
|
||||
verdicts: Final = {name: _is_model_cost_zero(model=name, llm_router=router) for name in order}
|
||||
|
||||
assert verdicts == expected
|
||||
assert {name: _is_model_cost_zero(model=name, llm_router=router) for name in order} == expected
|
||||
|
|
|
|||
|
|
@ -64,6 +64,7 @@ from litellm.types.router import (
|
|||
Deployment,
|
||||
DeploymentTypedDict,
|
||||
LiteLLM_Params,
|
||||
ModelGroupInfo,
|
||||
ModelInfo,
|
||||
PreRoutingHookResponse,
|
||||
RetryPolicy,
|
||||
|
|
@ -2333,6 +2334,116 @@ def test_update_settings_model_group_alias_drops_cached_group_info():
|
|||
assert after.input_cost_per_token is not None and after.input_cost_per_token > 0
|
||||
|
||||
|
||||
_PAID_INPUT_COST_PER_TOKEN: Final = 3e-06
|
||||
_PAID_OUTPUT_COST_PER_TOKEN: Final = 1.5e-05
|
||||
|
||||
|
||||
def _free_ollama_deployment(model_name: str) -> dict:
|
||||
return {
|
||||
"model_name": model_name,
|
||||
"litellm_params": {
|
||||
"model": "ollama/qwen3:0.6b",
|
||||
"api_base": "http://localhost:11434",
|
||||
"input_cost_per_token": 0,
|
||||
"output_cost_per_token": 0,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _paid_openai_deployment(model_name: str, model: str) -> dict:
|
||||
return {
|
||||
"model_name": model_name,
|
||||
"litellm_params": {
|
||||
"model": model,
|
||||
"api_key": "fake",
|
||||
"input_cost_per_token": _PAID_INPUT_COST_PER_TOKEN,
|
||||
"output_cost_per_token": _PAID_OUTPUT_COST_PER_TOKEN,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _assert_priced(info: ModelGroupInfo | None, provider: str) -> None:
|
||||
assert info is not None
|
||||
assert info.providers == [provider]
|
||||
assert info.input_cost_per_token == _PAID_INPUT_COST_PER_TOKEN
|
||||
assert info.output_cost_per_token == _PAID_OUTPUT_COST_PER_TOKEN
|
||||
|
||||
|
||||
def _assert_free(info: ModelGroupInfo | None, provider: str) -> None:
|
||||
assert info is not None
|
||||
assert info.providers == [provider]
|
||||
assert info.input_cost_per_token == 0
|
||||
assert info.output_cost_per_token == 0
|
||||
|
||||
|
||||
def test_get_model_group_info_prices_an_alias_chain_from_the_group_it_routes_to():
|
||||
"""chain-entry resolves one hop to local-free and is served by local-free's own
|
||||
deployment, so its price is that deployment's; local-free's own alias to gpt-priced
|
||||
is a hop the router takes only for a request to local-free by name."""
|
||||
router = Router(
|
||||
model_list=[
|
||||
_free_ollama_deployment("local-free"),
|
||||
_paid_openai_deployment("gpt-priced", "gpt-4o"),
|
||||
],
|
||||
model_group_alias={"chain-entry": "local-free", "local-free": "gpt-priced"},
|
||||
)
|
||||
|
||||
_assert_free(router.get_model_group_info(model_group="chain-entry"), "ollama")
|
||||
_assert_priced(router.get_model_group_info(model_group="local-free"), "openai")
|
||||
|
||||
|
||||
def test_get_model_group_info_prices_an_alias_chain_from_the_wildcard_route_serving_it():
|
||||
"""When the routed group has no deployment of its own, the wildcard route matching it
|
||||
serves the request, so the price is the wildcard's and never the routed group's own alias
|
||||
target's."""
|
||||
router = Router(
|
||||
model_list=[
|
||||
_paid_openai_deployment("openai/*", "openai/*"),
|
||||
_free_ollama_deployment("local-free"),
|
||||
],
|
||||
model_group_alias={"wildcard-entry": "openai/gpt-4o", "openai/gpt-4o": "local-free"},
|
||||
)
|
||||
|
||||
_assert_priced(router.get_model_group_info(model_group="wildcard-entry"), "openai")
|
||||
_assert_free(router.get_model_group_info(model_group="openai/gpt-4o"), "ollama")
|
||||
|
||||
|
||||
def _served_models(deployments: list[DeploymentTypedDict] | None) -> list[str]:
|
||||
return [deployment["litellm_params"]["model"] for deployment in deployments or ()]
|
||||
|
||||
|
||||
def test_get_model_list_of_routed_group_reads_the_groups_own_deployments_only():
|
||||
"""The router resolves an alias once, so a group reached as an alias target is served by
|
||||
its own deployments. get_model_list composes the group's own alias target too, the hop a
|
||||
request to that group by name takes."""
|
||||
router = Router(
|
||||
model_list=[
|
||||
_free_ollama_deployment("local-free"),
|
||||
_paid_openai_deployment("gpt-priced", "gpt-4o"),
|
||||
],
|
||||
model_group_alias={"local-free": "gpt-priced"},
|
||||
)
|
||||
|
||||
assert _served_models(router.get_model_list_of_routed_group("local-free")) == ["ollama/qwen3:0.6b"]
|
||||
assert _served_models(router.get_model_list(model_name="local-free")) == ["ollama/qwen3:0.6b", "gpt-4o"]
|
||||
|
||||
|
||||
def test_get_model_list_of_routed_group_falls_back_to_the_wildcard_route_serving_it():
|
||||
router = Router(
|
||||
model_list=[
|
||||
_paid_openai_deployment("openai/*", "openai/*"),
|
||||
_free_ollama_deployment("local-free"),
|
||||
],
|
||||
model_group_alias={"openai/gpt-4o": "local-free"},
|
||||
)
|
||||
|
||||
routed = router.get_model_list_of_routed_group("openai/gpt-4o")
|
||||
|
||||
assert [deployment["model_name"] for deployment in routed] == ["openai/gpt-4o"]
|
||||
assert _served_models(routed) == ["openai/gpt-4o"]
|
||||
assert _served_models(router.get_model_list(model_name="openai/gpt-4o")) == ["ollama/qwen3:0.6b"]
|
||||
|
||||
|
||||
def test_switch_routing_strategy_installs_lar1_then_restores_the_default_selector():
|
||||
router = _alias_cost_router()
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue