From ef284bbaca0cf3482445335eca64b58853e311b2 Mon Sep 17 00:00:00 2001 From: haydster7 Date: Fri, 18 Sep 2026 12:27:40 +1000 Subject: [PATCH] fix(proxy): resolve model aliases to their target deployment for listing metadata A model alias (a key's `aliases`, a team's `model_aliases`, or the router's `model_group_alias`) is rewritten at request time and owns no deployment row, so `get_model_listing_info` returns None for it and the alias name is not a cost-map key either. Both metadata candidate sources come up empty and the listing entry carries no `mode`, `max_input_tokens`, or `max_output_tokens`, while the deployment it routes to reports them on the same response. Resolve a listed alias to the deployment it routes to for the metadata lookup, keeping the alias as the response id. Maps are applied in order, each against the result of the last, mirroring how `litellm_pre_call_utils` applies the team map and then the key map, so a chained alias reports the deployment the caller actually reaches rather than an intermediate one. An alias shadowing a same-named deployment follows the alias, because the request rewrite does too. Fixes the model_aliases half of #40553. --- .../proxy/common_utils/model_listing_utils.py | 58 +++++ litellm/proxy/proxy_server.py | 25 +- .../test_team_model_name_translation.py | 234 ++++++++++++++++++ 3 files changed, 313 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/common_utils/model_listing_utils.py b/litellm/proxy/common_utils/model_listing_utils.py index 4b3ff2711b3..b5972909341 100644 --- a/litellm/proxy/common_utils/model_listing_utils.py +++ b/litellm/proxy/common_utils/model_listing_utils.py @@ -255,6 +255,64 @@ class TeamModelNameTranslator: return [(name, name) for name in model_names] return list(TeamModelNameTranslator._response_to_lookup_map(model_names, internal_to_public).items()) + @staticmethod + def resolve_alias_target( + listed_name: str, + llm_router: Router | None, + alias_maps: Sequence[object] = (), + ) -> str | None: + """The deployment `listed_name` routes to when it is an alias, else None. + + An alias owns no deployment row, so a metadata lookup keyed on it finds neither + a deployment nor a cost-map entry and yields no `mode` or token limits. + + Rewrites are applied in `alias_maps` order, each against the result of the last, + because `litellm_pre_call_utils` applies the team map and then the key map to + whatever name survived the previous one. Checking every map against + `listed_name` instead would report an intermediate deployment's metadata for a + chained alias. + """ + if llm_router is None: + return None + + resolved: Final = TeamModelNameTranslator._chain_alias_rewrites( + listed_name, (*alias_maps, llm_router.model_group_alias) + ) + if resolved == listed_name: + return None + return resolved if llm_router.model_name_to_deployment_indices.get(resolved) else None + + @staticmethod + def _chain_alias_rewrites(listed_name: str, alias_maps: Sequence[object]) -> str: + """`listed_name` after each map in turn rewrites the previous result. + + A name already rewritten once is never followed again, so a cycle across maps + terminates instead of looping. + """ + resolved = listed_name # mutable-ok: fold over the alias maps + seen: Final[set[str]] = {listed_name} # mutable-ok: accumulates visited names to break alias cycles + for alias_map in alias_maps: + target: str | None = TeamModelNameTranslator._alias_lookup(alias_map, resolved) + if target is None or target in seen: + continue + resolved = target + seen.add(target) + return resolved + + @staticmethod + def _alias_lookup(alias_map: object, name: str) -> str | None: + """`name`'s target in one alias map, or None when absent or malformed. + + `model_group_alias` entries may be a routing dict rather than a bare name. + """ + if not isinstance(alias_map, Mapping): + return None + raw_target: Final[object] = cast("Mapping[str, object]", alias_map).get(name) + target: Final[object] = ( + cast("Mapping[str, object]", raw_target).get("model") if isinstance(raw_target, Mapping) else raw_target + ) + return target if isinstance(target, str) and target else None + @staticmethod def translate_listing( model_names: list[str], diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index d7d8413d2ce..0766a31b905 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -10716,6 +10716,13 @@ async def model_list( view_aliases, ), ) + # Team map before key map, the order litellm_pre_call_utils rewrites in, so a + # chained alias resolves to the deployment a request would actually reach. + alias_maps: Final[Sequence[object]] = ( + user_api_key_dict.team_model_aliases, + user_api_key_dict.aliases, + view_aliases, + ) # Validate scope parameter if provided if scope is not None and scope != "expand": @@ -10788,7 +10795,7 @@ async def model_list( admin_entries: Final = TeamModelNameTranslator.listing_entries(all_models, llm_router, settings) for response_id, lookup_id in admin_entries: model_info = create_model_info_response( - model_id=lookup_id, + model_id=TeamModelNameTranslator.resolve_alias_target(lookup_id, llm_router, alias_maps) or lookup_id, provider="openai", include_metadata=include_metadata or False, fallback_type=fallback_type, @@ -10841,7 +10848,7 @@ async def model_list( entries: Final = TeamModelNameTranslator.listing_entries(all_models, llm_router, settings) for response_id, lookup_id in entries: model_info = create_model_info_response( - model_id=lookup_id, + model_id=TeamModelNameTranslator.resolve_alias_target(lookup_id, llm_router, alias_maps) or lookup_id, provider="openai", include_metadata=include_metadata or False, fallback_type=fallback_type, @@ -10948,7 +10955,17 @@ async def model_info( if llm_router is None: raise HTTPException(status_code=500, detail="Router not initialized") - deployment: Final = llm_router.get_deployment_by_model_group_name(resolved_model_id) + # An alias has no deployment row of its own, so resolve it to the deployment the + # request would actually reach; both the provider and the capability metadata + # belong to that model, while the response id stays the name the caller asked for. + alias_target: Final = TeamModelNameTranslator.resolve_alias_target( + resolved_model_id, + llm_router, + (user_api_key_dict.team_model_aliases, user_api_key_dict.aliases), + ) + deployment_lookup_id: Final = alias_target or resolved_model_id + + deployment: Final = llm_router.get_deployment_by_model_group_name(deployment_lookup_id) if deployment is None: raise HTTPException( status_code=404, @@ -10959,7 +10976,7 @@ async def model_info( _, provider, _, _ = litellm.get_llm_provider(model=deployment.litellm_params.model) response: Final = create_model_info_response( - model_id=resolved_model_id, + model_id=deployment_lookup_id, provider=provider, include_metadata=False, fallback_type=None, diff --git a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py index baa032f75e6..c66e0c22869 100644 --- a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py +++ b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py @@ -10,10 +10,12 @@ rows instead of the internal routing key `model_name_{team_id}_{uuid}`. from __future__ import annotations import json +from typing import Final from unittest.mock import AsyncMock, MagicMock import pytest +import litellm import litellm.proxy.proxy_server as ps from litellm.proxy._types import ( LiteLLM_UserTable, @@ -1893,3 +1895,235 @@ async def test_populate_team_access_grants_all_proxy_models_user_direct_access( assert [m["model_info"]["id"] for m in visible] == ["global-id-1"] assert visible[0]["model_info"]["direct_access"] is True + + +# --------------------------------------------------------------------------- +# Alias -> target metadata resolution (follow-up to #40553) +# +# A model alias (key `aliases`, team `model_aliases`, router `model_group_alias`) +# is rewritten at request time and owns no deployment row, so a metadata lookup +# keyed on the alias finds neither a deployment nor a cost-map entry and the +# listing entry carries no `mode` or token limits. The listed name must resolve +# to its target deployment for metadata while remaining the response id. +# --------------------------------------------------------------------------- + + +def _aliasing_router(deployment_names: tuple[str, ...], alias_map: dict) -> MagicMock: + """Router serving `deployment_names`, with `alias_map` as model_group_alias.""" + router: Final = MagicMock() + router.model_name_to_deployment_indices = {name: [i] for i, name in enumerate(deployment_names)} + router.model_group_alias = alias_map + return router + + +def test_resolve_alias_target_returns_alias_target(): + """An alias resolves to the deployment it points at.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target("claude-default", router, ({"claude-default": "claude-opus"},)) + == "claude-opus" + ) + + +def test_resolve_alias_target_follows_alias_shadowing_a_deployment(): + """An alias wins over a same-named deployment, matching the request rewrite. + + litellm_pre_call_utils rewrites the model name whenever the alias map has the + key, without checking whether that name is also a deployment, so the listing + has to report the target's metadata or it describes a model the caller never + reaches. + """ + router: Final = _aliasing_router(("claude-default", "claude-opus"), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target("claude-default", router, ({"claude-default": "claude-opus"},)) + == "claude-opus" + ) + + +def test_resolve_alias_target_returns_none_for_unresolvable_target(): + """An alias pointing at nothing the router serves falls back to the listed name.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert TeamModelNameTranslator.resolve_alias_target("ghost", router, ({"ghost": "not-deployed"},)) is None + + +def test_resolve_alias_target_ignores_self_referential_alias(): + """A self-referential alias must not be reported as its own target.""" + router: Final = _aliasing_router(("gpt-4o-mini",), {}) + assert TeamModelNameTranslator.resolve_alias_target("loop", router, ({"loop": "loop"},)) is None + + +def test_resolve_alias_target_applies_alias_maps_sequentially(): + """Each map rewrites the result of the last, as the request path does. + + The team map turns `default` into `mid`, then the key map turns `mid` into + `claude-opus`, so the final deployment is what the caller reaches. + """ + router: Final = _aliasing_router(("mid", "claude-opus"), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target( + "default", + router, + ({"default": "mid"}, {"mid": "claude-opus"}), + ) + == "claude-opus" + ) + + +def test_resolve_alias_target_stops_on_alias_cycle(): + """A cycle across maps terminates instead of looping.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target( + "a", + router, + ({"a": "b"}, {"b": "a"}), + ) + is None + ) + + +def test_resolve_alias_target_reads_router_model_group_alias(): + """Config-level model_group_alias is consulted after the caller's own maps.""" + router: Final = _aliasing_router(("claude-opus",), {"claude-default": "claude-opus"}) + assert TeamModelNameTranslator.resolve_alias_target("claude-default", router, ()) == "claude-opus" + + +def test_resolve_alias_target_unwraps_model_group_alias_dict(): + """model_group_alias entries may be a routing dict rather than a bare name.""" + router: Final = _aliasing_router(("claude-opus",), {"claude-default": {"model": "claude-opus", "hidden": True}}) + assert TeamModelNameTranslator.resolve_alias_target("claude-default", router, ()) == "claude-opus" + + +def test_resolve_alias_target_tolerates_non_mapping_alias_maps(): + """A malformed alias map is skipped rather than failing the listing.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert TeamModelNameTranslator.resolve_alias_target( + "claude-default", router, (None, "nonsense", {"claude-default": "claude-opus"}) + ) == ("claude-opus") + + +def test_resolve_alias_target_without_router(): + """No router means no deployments to resolve against.""" + assert TeamModelNameTranslator.resolve_alias_target("claude-default", None, ({"claude-default": "x"},)) is None + + +@pytest.mark.asyncio +async def test_v1_models_chained_team_then_key_alias_uses_request_order(monkeypatch): + """A chain spanning both maps resolves the way the request path rewrites it. + + The team map turns `chained` into `mid`, then the key map turns `mid` into + `claude-opus`. litellm_pre_call_utils applies the team map first, so the + caller reaches `claude-opus`; resolving key-first would stop at `mid` and + report a deployment the request never hits. + """ + opus: Final = { + "model_name": "claude-opus", + "litellm_params": {"model": "anthropic/claude-opus-4-20250514"}, + "model_info": {"id": "dep-opus"}, + } + mid: Final = { + "model_name": "mid", + "litellm_params": {"model": "openai/gpt-4o-mini"}, + "model_info": {"id": "dep-mid"}, + } + router: Final = MagicMock() + router.get_fully_blocked_model_names.return_value = set() + router.model_list = [opus, mid] + router.get_model_list.return_value = [opus, mid] + router.get_model_names.return_value = ["claude-opus", "mid"] + router.get_model_access_groups.return_value = {} + router.model_name_to_deployment_indices = {"claude-opus": [0], "mid": [1]} + router.model_group_alias = {} + router.get_configured_mode.return_value = None + router.get_configured_token_limits.return_value = (None, None) + listing: Final = { + "claude-opus": DeploymentModelListingInfo( + cost_map_keys=("anthropic/claude-opus-4-20250514",), + max_input_tokens=None, + max_output_tokens=None, + ), + "mid": DeploymentModelListingInfo( + cost_map_keys=("openai/gpt-4o-mini",), + max_input_tokens=None, + max_output_tokens=None, + ), + } + router.get_model_listing_info.side_effect = listing.get + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {}) + + key: Final = UserAPIKeyAuth( + user_id="u", + api_key="sk-test", + models=[], + team_models=["chained", "claude-opus", "mid"], + team_model_aliases={"chained": "mid"}, + aliases={"mid": "claude-opus"}, + ) + resp: Final = await ps.model_list(user_api_key_dict=key) + + rows: Final = {d["id"]: d for d in resp["data"]} + opus_limits: Final = (rows["claude-opus"]["mode"], rows["claude-opus"]["max_input_tokens"]) + mid_limits: Final = litellm.get_model_info("openai/gpt-4o-mini") + assert (rows["chained"]["mode"], rows["chained"]["max_input_tokens"]) == opus_limits + # Stopping at the intermediate deployment would report gpt-4o-mini's window. + assert rows["chained"]["max_input_tokens"] != mid_limits["max_input_tokens"] + + +@pytest.mark.asyncio +async def test_v1_models_alias_reports_target_deployment_metadata(monkeypatch): + """Regression (#40553 follow-up): an alias listed in the team's models array + must carry the mode and token limits of the deployment it resolves to. + + Before the fix the alias listed with no metadata while its target reported + full metadata on the same response. + """ + deployment: Final = { + "model_name": "claude-opus", + "litellm_params": {"model": "anthropic/claude-opus-4-20250514"}, + "model_info": {"id": "dep-1"}, + } + router: Final = MagicMock() + router.get_fully_blocked_model_names.return_value = set() + router.model_list = [deployment] + router.get_model_list.return_value = [deployment] + router.get_model_names.return_value = ["claude-opus"] + router.get_model_access_groups.return_value = {} + router.model_name_to_deployment_indices = {"claude-opus": [0]} + router.model_group_alias = {} + router.get_configured_mode.return_value = None + router.get_configured_token_limits.return_value = (None, None) + # Only the real deployment has listing info; the alias is not a deployment. + router.get_model_listing_info.side_effect = lambda name: ( + DeploymentModelListingInfo( + cost_map_keys=("anthropic/claude-opus-4-20250514",), + max_input_tokens=None, + max_output_tokens=None, + ) + if name == "claude-opus" + else None + ) + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {}) + + # The alias name is in the team's models array for discoverability, and in + # model_aliases for routing. + key: Final = UserAPIKeyAuth( + user_id="u", + api_key="sk-test", + models=[], + team_models=["claude-default", "claude-opus"], + team_model_aliases={"claude-default": "claude-opus"}, + ) + resp: Final = await ps.model_list(user_api_key_dict=key) + + rows: Final = {d["id"]: d for d in resp["data"]} + assert "claude-default" in rows, "the alias must remain listed under its own name" + assert rows["claude-default"].get("mode") == "chat" + # The alias reports the same capability metadata as its target deployment. + assert rows["claude-default"].get("max_input_tokens") == rows["claude-opus"].get("max_input_tokens") + assert rows["claude-default"]["max_input_tokens"] is not None