fix(router): price a model group from the deployments that serve it (#44732)

* fix(router): price a model group from the deployments that serve it

A model group's info read composed the group's own model_group_alias
entry into its deployments, a hop the router never takes: an alias is
resolved exactly once at request time, so a group reached as an alias
target is served by its own deployments. For the chain X -> T -> U the
price read for X included U's deployments too, and the free-model
budget waiver refused a free request to X on an over-budget key, while
GET /model_group/info reported U's providers and price for X.

The group info read now prices a group from the deployments routing
serves it with: the ones named after it, the routing group of that
name, or the wildcard route matching it when neither exists. The
budget waiver, GET /model_group/info, the rate limiters, and the
response headers all read the same set as routing. get_model_list
keeps its behavior for every other caller.

* test(router): give the paid fixtures explicit per-token prices

* test(router): call the routed-group read by name so the router coverage gate sees it

* test(integration): audit the alias chain budget waiver on every route, shape, and outage

Thirty-four cells under the management group prove an over-budget key is served through an alias chain entry at the price of the deployment that serves it, on chat, responses, and messages, sync and streamed, through the OpenAI and Anthropic SDKs and raw httpx on both replicas, with the chain middle, the reverse chain, a cost-map priced middle, a ghost middle, wildcard and routing-group targets, malformed and hostile inputs, a cached reply, a repointed alias, a provider failure, and two chaos bursts (a killed worker, a scripted outage)

---------

Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-06 05:31:09 +00:00 • committed by GitHub
parent c2c0bb583e
commit c13746c951
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 1204 additions and 18 deletions

View file

@ -11276,9 +11276,7 @@ class Router:
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None
reasoning_efforts_initialized = False
reasoning_efforts_unknown = False
model_list: Final = self.get_model_list(model_name=model_group)
if model_list is None:
return None
model_list: Final = self.get_model_list_of_routed_group(model_group)
for model in model_list:
is_match = False
if (
@ -12409,27 +12407,43 @@ class Router:
returned_models.extend(self.get_model_list_from_model_alias(model_name=model_name))
returned_models.extend(self.get_model_list_from_routing_groups(model_name=model_name))
if len(returned_models) == 0: # check if wildcard route
potential_wildcard_models: Final = self.pattern_router.get_deployments_by_pattern(model=model_name or "")
## check for team-specific wildcard models
if team_id is not None and team_id in self.team_pattern_routers:
potential_team_only_wildcard_models: Final = self.team_pattern_routers[
team_id
].get_deployments_by_pattern(model=model_name or "")
potential_wildcard_models.extend(potential_team_only_wildcard_models)
if model_name is not None and potential_wildcard_models is not None:
for m in potential_wildcard_models:
deployment_typed_dict = DeploymentTypedDict(**m)
deployment_typed_dict["model_name"] = model_name
returned_models.append(deployment_typed_dict)
if len(returned_models) == 0 and model_name is not None:
returned_models.extend(self._get_wildcard_deployments(model_name=model_name, team_id=team_id))
if model_name is None:
returned_models += self.model_list
return returned_models
def _get_wildcard_deployments(self, model_name: str, team_id: str | None = None) -> list[DeploymentTypedDict]:
"""
The deployments of the wildcard routes matching model_name (the proxy-wide
ones, plus team_id's own when given), each emitted under model_name.
"""
team_router: Final = self.team_pattern_routers.get(team_id) if team_id is not None else None
matches: Final = [
*self.pattern_router.get_deployments_by_pattern(model=model_name),
*(team_router.get_deployments_by_pattern(model=model_name) if team_router is not None else ()),
]
return [{**DeploymentTypedDict(**m), "model_name": model_name} for m in matches]
def get_model_list_of_routed_group(self, model_group: str) -> list[DeploymentTypedDict]:
"""
The deployments a request the router has already resolved to model_group is
served from: the ones named model_group, the routing group of that name, or
the wildcard route matching it when neither exists.
Unlike get_model_list, model_group's own model_group_alias entry is not
followed. The router resolves an alias exactly once, so a group reached as
an alias target is served by its own deployments, never by a second hop:
in the chain X -> T -> U a request to X is served from T's deployments.
"""
named: Final = [
*self._get_all_deployments(model_name=model_group),
*self.get_model_list_from_routing_groups(model_name=model_group),
]
return named or self._get_wildcard_deployments(model_name=model_group)
def resolved_litellm_models(self, model_name: str, team_id: str | None = None) -> tuple[str, ...]:
"""The provider model strings `model_name` can actually be served by on this proxy.

View file

@ -105,6 +105,7 @@ ignored_function_names = [
"_aanthropic_messages_retry_same_group", # Tested through the dropped-before-content retry tests in test_router.py
"_aanthropic_messages_yield_recovered", # Tested through every mid-stream retry and fallback test in test_router.py
"_anthropic_messages_policy_retries", # Tested through the retry budget precedence test in test_router.py
"_get_wildcard_deployments", # Tested through the get_model_list_of_routed_group wildcard test in test_router.py
]

File diff suppressed because it is too large Load diff

View file

@ -32,6 +32,7 @@ from litellm.proxy._types import (
from litellm.proxy.utils import PrismaClient
from litellm.proxy.auth.auth_checks import (
can_team_access_model,
_is_model_cost_zero,
_virtual_key_soft_budget_check,
_team_soft_budget_check,
)
@ -1595,3 +1596,44 @@ async def test_get_user_object_cache_miss_emits_exactly_one_postgres_get_user_ob
assert result is not None and result.user_id == user_id
assert await _db_service_call_types(db_success_hook) == ("get_user_object",)
assert db_success_hook.await_args_list[0].kwargs["parent_otel_span"] == "auth-span"
@pytest.mark.parametrize("entry_first", [True, False])
def test_is_model_cost_zero_judges_an_alias_chain_by_the_deployment_its_entry_routes_to(
monkeypatch: pytest.MonkeyPatch, entry_first: bool
) -> None:
"""chain-entry resolves one hop to local-free and is served by local-free's own free
deployment, so an over-budget key is waived for it; local-free by name resolves to paid-gpt
and stays enforced. local-free's alias is a hop the router never takes for chain-entry, and
the per-name verdict cache must not let either name's verdict leak into the other's."""
monkeypatch.setattr(litellm, "model_cost", dict(litellm.model_cost))
router: Final = litellm.Router(
model_list=[
{
"model_name": "local-free",
"litellm_params": {
"model": "ollama/qwen3:0.6b",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0,
"output_cost_per_token": 0,
},
},
{
"model_name": "paid-gpt",
"litellm_params": {
"model": "gpt-4o",
"api_key": "fake",
"input_cost_per_token": 3e-06,
"output_cost_per_token": 1.5e-05,
},
},
],
model_group_alias={"chain-entry": "local-free", "local-free": "paid-gpt"},
)
expected: Final = {"chain-entry": True, "local-free": False, "paid-gpt": False}
order: Final = ("chain-entry", "local-free", "paid-gpt") if entry_first else ("local-free", "paid-gpt", "chain-entry")
verdicts: Final = {name: _is_model_cost_zero(model=name, llm_router=router) for name in order}
assert verdicts == expected
assert {name: _is_model_cost_zero(model=name, llm_router=router) for name in order} == expected

View file

@ -64,6 +64,7 @@ from litellm.types.router import (
Deployment,
DeploymentTypedDict,
LiteLLM_Params,
ModelGroupInfo,
ModelInfo,
PreRoutingHookResponse,
RetryPolicy,
@ -2333,6 +2334,116 @@ def test_update_settings_model_group_alias_drops_cached_group_info():
assert after.input_cost_per_token is not None and after.input_cost_per_token > 0
_PAID_INPUT_COST_PER_TOKEN: Final = 3e-06
_PAID_OUTPUT_COST_PER_TOKEN: Final = 1.5e-05
def _free_ollama_deployment(model_name: str) -> dict:
return {
"model_name": model_name,
"litellm_params": {
"model": "ollama/qwen3:0.6b",
"api_base": "http://localhost:11434",
"input_cost_per_token": 0,
"output_cost_per_token": 0,
},
}
def _paid_openai_deployment(model_name: str, model: str) -> dict:
return {
"model_name": model_name,
"litellm_params": {
"model": model,
"api_key": "fake",
"input_cost_per_token": _PAID_INPUT_COST_PER_TOKEN,
"output_cost_per_token": _PAID_OUTPUT_COST_PER_TOKEN,
},
}
def _assert_priced(info: ModelGroupInfo | None, provider: str) -> None:
assert info is not None
assert info.providers == [provider]
assert info.input_cost_per_token == _PAID_INPUT_COST_PER_TOKEN
assert info.output_cost_per_token == _PAID_OUTPUT_COST_PER_TOKEN
def _assert_free(info: ModelGroupInfo | None, provider: str) -> None:
assert info is not None
assert info.providers == [provider]
assert info.input_cost_per_token == 0
assert info.output_cost_per_token == 0
def test_get_model_group_info_prices_an_alias_chain_from_the_group_it_routes_to():
"""chain-entry resolves one hop to local-free and is served by local-free's own
deployment, so its price is that deployment's; local-free's own alias to gpt-priced
is a hop the router takes only for a request to local-free by name."""
router = Router(
model_list=[
_free_ollama_deployment("local-free"),
_paid_openai_deployment("gpt-priced", "gpt-4o"),
],
model_group_alias={"chain-entry": "local-free", "local-free": "gpt-priced"},
)
_assert_free(router.get_model_group_info(model_group="chain-entry"), "ollama")
_assert_priced(router.get_model_group_info(model_group="local-free"), "openai")
def test_get_model_group_info_prices_an_alias_chain_from_the_wildcard_route_serving_it():
"""When the routed group has no deployment of its own, the wildcard route matching it
serves the request, so the price is the wildcard's and never the routed group's own alias
target's."""
router = Router(
model_list=[
_paid_openai_deployment("openai/*", "openai/*"),
_free_ollama_deployment("local-free"),
],
model_group_alias={"wildcard-entry": "openai/gpt-4o", "openai/gpt-4o": "local-free"},
)
_assert_priced(router.get_model_group_info(model_group="wildcard-entry"), "openai")
_assert_free(router.get_model_group_info(model_group="openai/gpt-4o"), "ollama")
def _served_models(deployments: list[DeploymentTypedDict] | None) -> list[str]:
return [deployment["litellm_params"]["model"] for deployment in deployments or ()]
def test_get_model_list_of_routed_group_reads_the_groups_own_deployments_only():
"""The router resolves an alias once, so a group reached as an alias target is served by
its own deployments. get_model_list composes the group's own alias target too, the hop a
request to that group by name takes."""
router = Router(
model_list=[
_free_ollama_deployment("local-free"),
_paid_openai_deployment("gpt-priced", "gpt-4o"),
],
model_group_alias={"local-free": "gpt-priced"},
)
assert _served_models(router.get_model_list_of_routed_group("local-free")) == ["ollama/qwen3:0.6b"]
assert _served_models(router.get_model_list(model_name="local-free")) == ["ollama/qwen3:0.6b", "gpt-4o"]
def test_get_model_list_of_routed_group_falls_back_to_the_wildcard_route_serving_it():
router = Router(
model_list=[
_paid_openai_deployment("openai/*", "openai/*"),
_free_ollama_deployment("local-free"),
],
model_group_alias={"openai/gpt-4o": "local-free"},
)
routed = router.get_model_list_of_routed_group("openai/gpt-4o")
assert [deployment["model_name"] for deployment in routed] == ["openai/gpt-4o"]
assert _served_models(routed) == ["openai/gpt-4o"]
assert _served_models(router.get_model_list(model_name="openai/gpt-4o")) == ["ollama/qwen3:0.6b"]
def test_switch_routing_strategy_installs_lar1_then_restores_the_default_selector():
router = _alias_cost_router()