diff --git a/litellm/proxy/common_utils/model_listing_utils.py b/litellm/proxy/common_utils/model_listing_utils.py index 4b3ff2711b3..b5972909341 100644 --- a/litellm/proxy/common_utils/model_listing_utils.py +++ b/litellm/proxy/common_utils/model_listing_utils.py @@ -255,6 +255,64 @@ class TeamModelNameTranslator: return [(name, name) for name in model_names] return list(TeamModelNameTranslator._response_to_lookup_map(model_names, internal_to_public).items()) + @staticmethod + def resolve_alias_target( + listed_name: str, + llm_router: Router | None, + alias_maps: Sequence[object] = (), + ) -> str | None: + """The deployment `listed_name` routes to when it is an alias, else None. + + An alias owns no deployment row, so a metadata lookup keyed on it finds neither + a deployment nor a cost-map entry and yields no `mode` or token limits. + + Rewrites are applied in `alias_maps` order, each against the result of the last, + because `litellm_pre_call_utils` applies the team map and then the key map to + whatever name survived the previous one. Checking every map against + `listed_name` instead would report an intermediate deployment's metadata for a + chained alias. + """ + if llm_router is None: + return None + + resolved: Final = TeamModelNameTranslator._chain_alias_rewrites( + listed_name, (*alias_maps, llm_router.model_group_alias) + ) + if resolved == listed_name: + return None + return resolved if llm_router.model_name_to_deployment_indices.get(resolved) else None + + @staticmethod + def _chain_alias_rewrites(listed_name: str, alias_maps: Sequence[object]) -> str: + """`listed_name` after each map in turn rewrites the previous result. + + A name already rewritten once is never followed again, so a cycle across maps + terminates instead of looping. + """ + resolved = listed_name # mutable-ok: fold over the alias maps + seen: Final[set[str]] = {listed_name} # mutable-ok: accumulates visited names to break alias cycles + for alias_map in alias_maps: + target: str | None = TeamModelNameTranslator._alias_lookup(alias_map, resolved) + if target is None or target in seen: + continue + resolved = target + seen.add(target) + return resolved + + @staticmethod + def _alias_lookup(alias_map: object, name: str) -> str | None: + """`name`'s target in one alias map, or None when absent or malformed. + + `model_group_alias` entries may be a routing dict rather than a bare name. + """ + if not isinstance(alias_map, Mapping): + return None + raw_target: Final[object] = cast("Mapping[str, object]", alias_map).get(name) + target: Final[object] = ( + cast("Mapping[str, object]", raw_target).get("model") if isinstance(raw_target, Mapping) else raw_target + ) + return target if isinstance(target, str) and target else None + @staticmethod def translate_listing( model_names: list[str], diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index d7d8413d2ce..0766a31b905 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -10716,6 +10716,13 @@ async def model_list( view_aliases, ), ) + # Team map before key map, the order litellm_pre_call_utils rewrites in, so a + # chained alias resolves to the deployment a request would actually reach. + alias_maps: Final[Sequence[object]] = ( + user_api_key_dict.team_model_aliases, + user_api_key_dict.aliases, + view_aliases, + ) # Validate scope parameter if provided if scope is not None and scope != "expand": @@ -10788,7 +10795,7 @@ async def model_list( admin_entries: Final = TeamModelNameTranslator.listing_entries(all_models, llm_router, settings) for response_id, lookup_id in admin_entries: model_info = create_model_info_response( - model_id=lookup_id, + model_id=TeamModelNameTranslator.resolve_alias_target(lookup_id, llm_router, alias_maps) or lookup_id, provider="openai", include_metadata=include_metadata or False, fallback_type=fallback_type, @@ -10841,7 +10848,7 @@ async def model_list( entries: Final = TeamModelNameTranslator.listing_entries(all_models, llm_router, settings) for response_id, lookup_id in entries: model_info = create_model_info_response( - model_id=lookup_id, + model_id=TeamModelNameTranslator.resolve_alias_target(lookup_id, llm_router, alias_maps) or lookup_id, provider="openai", include_metadata=include_metadata or False, fallback_type=fallback_type, @@ -10948,7 +10955,17 @@ async def model_info( if llm_router is None: raise HTTPException(status_code=500, detail="Router not initialized") - deployment: Final = llm_router.get_deployment_by_model_group_name(resolved_model_id) + # An alias has no deployment row of its own, so resolve it to the deployment the + # request would actually reach; both the provider and the capability metadata + # belong to that model, while the response id stays the name the caller asked for. + alias_target: Final = TeamModelNameTranslator.resolve_alias_target( + resolved_model_id, + llm_router, + (user_api_key_dict.team_model_aliases, user_api_key_dict.aliases), + ) + deployment_lookup_id: Final = alias_target or resolved_model_id + + deployment: Final = llm_router.get_deployment_by_model_group_name(deployment_lookup_id) if deployment is None: raise HTTPException( status_code=404, @@ -10959,7 +10976,7 @@ async def model_info( _, provider, _, _ = litellm.get_llm_provider(model=deployment.litellm_params.model) response: Final = create_model_info_response( - model_id=resolved_model_id, + model_id=deployment_lookup_id, provider=provider, include_metadata=False, fallback_type=None, diff --git a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py index baa032f75e6..c66e0c22869 100644 --- a/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py +++ b/tests/test_litellm/proxy/proxy_server/test_team_model_name_translation.py @@ -10,10 +10,12 @@ rows instead of the internal routing key `model_name_{team_id}_{uuid}`. from __future__ import annotations import json +from typing import Final from unittest.mock import AsyncMock, MagicMock import pytest +import litellm import litellm.proxy.proxy_server as ps from litellm.proxy._types import ( LiteLLM_UserTable, @@ -1893,3 +1895,235 @@ async def test_populate_team_access_grants_all_proxy_models_user_direct_access( assert [m["model_info"]["id"] for m in visible] == ["global-id-1"] assert visible[0]["model_info"]["direct_access"] is True + + +# --------------------------------------------------------------------------- +# Alias -> target metadata resolution (follow-up to #40553) +# +# A model alias (key `aliases`, team `model_aliases`, router `model_group_alias`) +# is rewritten at request time and owns no deployment row, so a metadata lookup +# keyed on the alias finds neither a deployment nor a cost-map entry and the +# listing entry carries no `mode` or token limits. The listed name must resolve +# to its target deployment for metadata while remaining the response id. +# --------------------------------------------------------------------------- + + +def _aliasing_router(deployment_names: tuple[str, ...], alias_map: dict) -> MagicMock: + """Router serving `deployment_names`, with `alias_map` as model_group_alias.""" + router: Final = MagicMock() + router.model_name_to_deployment_indices = {name: [i] for i, name in enumerate(deployment_names)} + router.model_group_alias = alias_map + return router + + +def test_resolve_alias_target_returns_alias_target(): + """An alias resolves to the deployment it points at.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target("claude-default", router, ({"claude-default": "claude-opus"},)) + == "claude-opus" + ) + + +def test_resolve_alias_target_follows_alias_shadowing_a_deployment(): + """An alias wins over a same-named deployment, matching the request rewrite. + + litellm_pre_call_utils rewrites the model name whenever the alias map has the + key, without checking whether that name is also a deployment, so the listing + has to report the target's metadata or it describes a model the caller never + reaches. + """ + router: Final = _aliasing_router(("claude-default", "claude-opus"), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target("claude-default", router, ({"claude-default": "claude-opus"},)) + == "claude-opus" + ) + + +def test_resolve_alias_target_returns_none_for_unresolvable_target(): + """An alias pointing at nothing the router serves falls back to the listed name.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert TeamModelNameTranslator.resolve_alias_target("ghost", router, ({"ghost": "not-deployed"},)) is None + + +def test_resolve_alias_target_ignores_self_referential_alias(): + """A self-referential alias must not be reported as its own target.""" + router: Final = _aliasing_router(("gpt-4o-mini",), {}) + assert TeamModelNameTranslator.resolve_alias_target("loop", router, ({"loop": "loop"},)) is None + + +def test_resolve_alias_target_applies_alias_maps_sequentially(): + """Each map rewrites the result of the last, as the request path does. + + The team map turns `default` into `mid`, then the key map turns `mid` into + `claude-opus`, so the final deployment is what the caller reaches. + """ + router: Final = _aliasing_router(("mid", "claude-opus"), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target( + "default", + router, + ({"default": "mid"}, {"mid": "claude-opus"}), + ) + == "claude-opus" + ) + + +def test_resolve_alias_target_stops_on_alias_cycle(): + """A cycle across maps terminates instead of looping.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert ( + TeamModelNameTranslator.resolve_alias_target( + "a", + router, + ({"a": "b"}, {"b": "a"}), + ) + is None + ) + + +def test_resolve_alias_target_reads_router_model_group_alias(): + """Config-level model_group_alias is consulted after the caller's own maps.""" + router: Final = _aliasing_router(("claude-opus",), {"claude-default": "claude-opus"}) + assert TeamModelNameTranslator.resolve_alias_target("claude-default", router, ()) == "claude-opus" + + +def test_resolve_alias_target_unwraps_model_group_alias_dict(): + """model_group_alias entries may be a routing dict rather than a bare name.""" + router: Final = _aliasing_router(("claude-opus",), {"claude-default": {"model": "claude-opus", "hidden": True}}) + assert TeamModelNameTranslator.resolve_alias_target("claude-default", router, ()) == "claude-opus" + + +def test_resolve_alias_target_tolerates_non_mapping_alias_maps(): + """A malformed alias map is skipped rather than failing the listing.""" + router: Final = _aliasing_router(("claude-opus",), {}) + assert TeamModelNameTranslator.resolve_alias_target( + "claude-default", router, (None, "nonsense", {"claude-default": "claude-opus"}) + ) == ("claude-opus") + + +def test_resolve_alias_target_without_router(): + """No router means no deployments to resolve against.""" + assert TeamModelNameTranslator.resolve_alias_target("claude-default", None, ({"claude-default": "x"},)) is None + + +@pytest.mark.asyncio +async def test_v1_models_chained_team_then_key_alias_uses_request_order(monkeypatch): + """A chain spanning both maps resolves the way the request path rewrites it. + + The team map turns `chained` into `mid`, then the key map turns `mid` into + `claude-opus`. litellm_pre_call_utils applies the team map first, so the + caller reaches `claude-opus`; resolving key-first would stop at `mid` and + report a deployment the request never hits. + """ + opus: Final = { + "model_name": "claude-opus", + "litellm_params": {"model": "anthropic/claude-opus-4-20250514"}, + "model_info": {"id": "dep-opus"}, + } + mid: Final = { + "model_name": "mid", + "litellm_params": {"model": "openai/gpt-4o-mini"}, + "model_info": {"id": "dep-mid"}, + } + router: Final = MagicMock() + router.get_fully_blocked_model_names.return_value = set() + router.model_list = [opus, mid] + router.get_model_list.return_value = [opus, mid] + router.get_model_names.return_value = ["claude-opus", "mid"] + router.get_model_access_groups.return_value = {} + router.model_name_to_deployment_indices = {"claude-opus": [0], "mid": [1]} + router.model_group_alias = {} + router.get_configured_mode.return_value = None + router.get_configured_token_limits.return_value = (None, None) + listing: Final = { + "claude-opus": DeploymentModelListingInfo( + cost_map_keys=("anthropic/claude-opus-4-20250514",), + max_input_tokens=None, + max_output_tokens=None, + ), + "mid": DeploymentModelListingInfo( + cost_map_keys=("openai/gpt-4o-mini",), + max_input_tokens=None, + max_output_tokens=None, + ), + } + router.get_model_listing_info.side_effect = listing.get + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {}) + + key: Final = UserAPIKeyAuth( + user_id="u", + api_key="sk-test", + models=[], + team_models=["chained", "claude-opus", "mid"], + team_model_aliases={"chained": "mid"}, + aliases={"mid": "claude-opus"}, + ) + resp: Final = await ps.model_list(user_api_key_dict=key) + + rows: Final = {d["id"]: d for d in resp["data"]} + opus_limits: Final = (rows["claude-opus"]["mode"], rows["claude-opus"]["max_input_tokens"]) + mid_limits: Final = litellm.get_model_info("openai/gpt-4o-mini") + assert (rows["chained"]["mode"], rows["chained"]["max_input_tokens"]) == opus_limits + # Stopping at the intermediate deployment would report gpt-4o-mini's window. + assert rows["chained"]["max_input_tokens"] != mid_limits["max_input_tokens"] + + +@pytest.mark.asyncio +async def test_v1_models_alias_reports_target_deployment_metadata(monkeypatch): + """Regression (#40553 follow-up): an alias listed in the team's models array + must carry the mode and token limits of the deployment it resolves to. + + Before the fix the alias listed with no metadata while its target reported + full metadata on the same response. + """ + deployment: Final = { + "model_name": "claude-opus", + "litellm_params": {"model": "anthropic/claude-opus-4-20250514"}, + "model_info": {"id": "dep-1"}, + } + router: Final = MagicMock() + router.get_fully_blocked_model_names.return_value = set() + router.model_list = [deployment] + router.get_model_list.return_value = [deployment] + router.get_model_names.return_value = ["claude-opus"] + router.get_model_access_groups.return_value = {} + router.model_name_to_deployment_indices = {"claude-opus": [0]} + router.model_group_alias = {} + router.get_configured_mode.return_value = None + router.get_configured_token_limits.return_value = (None, None) + # Only the real deployment has listing info; the alias is not a deployment. + router.get_model_listing_info.side_effect = lambda name: ( + DeploymentModelListingInfo( + cost_map_keys=("anthropic/claude-opus-4-20250514",), + max_input_tokens=None, + max_output_tokens=None, + ) + if name == "claude-opus" + else None + ) + + monkeypatch.setattr(ps, "llm_router", router) + monkeypatch.setattr(ps, "user_model", None) + monkeypatch.setattr(ps, "general_settings", {}) + + # The alias name is in the team's models array for discoverability, and in + # model_aliases for routing. + key: Final = UserAPIKeyAuth( + user_id="u", + api_key="sk-test", + models=[], + team_models=["claude-default", "claude-opus"], + team_model_aliases={"claude-default": "claude-opus"}, + ) + resp: Final = await ps.model_list(user_api_key_dict=key) + + rows: Final = {d["id"]: d for d in resp["data"]} + assert "claude-default" in rows, "the alias must remain listed under its own name" + assert rows["claude-default"].get("mode") == "chat" + # The alias reports the same capability metadata as its target deployment. + assert rows["claude-default"].get("max_input_tokens") == rows["claude-opus"].get("max_input_tokens") + assert rows["claude-default"]["max_input_tokens"] is not None