diff --git a/litellm/router.py b/litellm/router.py index 0b9f12c8da3..cfbc6e57806 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -10447,6 +10447,10 @@ class Router: configured or discovered token limits. Resolved via O(1) index lookup. + A ``model_group_alias`` name that is not itself a deployment name resolves to its + target group, as routing does, so a listed alias reports the limits of the + deployments it routes to rather than none at all. + Returns None for wildcard-expanded or unknown names, where the listed name is the real model name and no deployment-specific information exists, and treats a malformed configured limit as absent rather than failing the listing. @@ -10459,7 +10463,9 @@ class Router: this never triggers pattern matching or deep copies, so it is safe to call per listed model on the /v1/models hot path. """ - indices: Final = self.model_name_to_deployment_indices.get(model_name) + indices: Final = self.model_name_to_deployment_indices.get( + model_name + ) or self.model_name_to_deployment_indices.get(self._get_model_from_alias(model_name) or "") if not indices: return None diff --git a/tests/unit/test_router_model_listing_alias.py b/tests/unit/test_router_model_listing_alias.py new file mode 100644 index 00000000000..ef4a32483b9 --- /dev/null +++ b/tests/unit/test_router_model_listing_alias.py @@ -0,0 +1,52 @@ +""" +A model_group_alias listed by /v1/models reports the token limits of the group it +routes to. Before this, the listing lookup read only deployment names, so an alias +row carried no max_input_tokens while /model/info showed them, and Claude Code's +picker view could not mark a 1M-context alias with [1m]. +""" + +from typing import Final + +from litellm import Router +from litellm.proxy.utils import create_model_info_response + + +def _router(model_group_alias: dict) -> Router: + return Router( + model_list=[ + { + "model_name": "long-context", + "litellm_params": {"model": "openai/org/long-context-model", "api_key": "k"}, + "model_info": {"id": "lc", "max_input_tokens": 1_000_000, "max_output_tokens": 128_000}, + }, + { + "model_name": "short", + "litellm_params": {"model": "openai/org/short-model", "api_key": "k"}, + "model_info": {"id": "s", "max_input_tokens": 8192, "max_output_tokens": 4096}, + }, + ], + model_group_alias=model_group_alias, + ) + + +def test_alias_reports_its_target_groups_limits() -> None: + router: Final = _router({"big": "long-context", "item": {"model": "long-context", "hidden": False}}) + for alias in ("big", "item"): + listing = router.get_model_listing_info(alias) + assert listing is not None + assert (listing.max_input_tokens, listing.max_output_tokens) == (1_000_000, 128_000) + row = create_model_info_response(model_id=alias, provider="openai", llm_router=router) + assert (row["id"], row.get("max_input_tokens"), row.get("max_output_tokens")) == (alias, 1_000_000, 128_000) + + +def test_a_deployment_name_wins_over_an_alias_of_the_same_name() -> None: + router: Final = _router({"short": "long-context"}) + listing: Final = router.get_model_listing_info("short") + assert listing is not None + assert listing.max_input_tokens == 8192 + + +def test_alias_to_an_unknown_group_and_unknown_names_stay_none() -> None: + router: Final = _router({"dangling": "no-such-group"}) + assert router.get_model_listing_info("dangling") is None + assert router.get_model_listing_info("not-listed") is None