mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Merge e0aa0a093c into 635085ac14
This commit is contained in:
commit
3f5179a42c
2 changed files with 59 additions and 1 deletions
|
|
@ -10447,6 +10447,10 @@ class Router:
|
|||
configured or discovered token limits. Resolved via O(1) index
|
||||
lookup.
|
||||
|
||||
A ``model_group_alias`` name that is not itself a deployment name resolves to its
|
||||
target group, as routing does, so a listed alias reports the limits of the
|
||||
deployments it routes to rather than none at all.
|
||||
|
||||
Returns None for wildcard-expanded or unknown names, where the listed name is the
|
||||
real model name and no deployment-specific information exists, and treats a
|
||||
malformed configured limit as absent rather than failing the listing.
|
||||
|
|
@ -10459,7 +10463,9 @@ class Router:
|
|||
this never triggers pattern matching or deep copies, so it is safe to call per
|
||||
listed model on the /v1/models hot path.
|
||||
"""
|
||||
indices: Final = self.model_name_to_deployment_indices.get(model_name)
|
||||
indices: Final = self.model_name_to_deployment_indices.get(
|
||||
model_name
|
||||
) or self.model_name_to_deployment_indices.get(self._get_model_from_alias(model_name) or "")
|
||||
if not indices:
|
||||
return None
|
||||
|
||||
|
|
|
|||
52
tests/unit/test_router_model_listing_alias.py
Normal file
52
tests/unit/test_router_model_listing_alias.py
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
"""
|
||||
A model_group_alias listed by /v1/models reports the token limits of the group it
|
||||
routes to. Before this, the listing lookup read only deployment names, so an alias
|
||||
row carried no max_input_tokens while /model/info showed them, and Claude Code's
|
||||
picker view could not mark a 1M-context alias with [1m].
|
||||
"""
|
||||
|
||||
from typing import Final
|
||||
|
||||
from litellm import Router
|
||||
from litellm.proxy.utils import create_model_info_response
|
||||
|
||||
|
||||
def _router(model_group_alias: dict) -> Router:
|
||||
return Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "long-context",
|
||||
"litellm_params": {"model": "openai/org/long-context-model", "api_key": "k"},
|
||||
"model_info": {"id": "lc", "max_input_tokens": 1_000_000, "max_output_tokens": 128_000},
|
||||
},
|
||||
{
|
||||
"model_name": "short",
|
||||
"litellm_params": {"model": "openai/org/short-model", "api_key": "k"},
|
||||
"model_info": {"id": "s", "max_input_tokens": 8192, "max_output_tokens": 4096},
|
||||
},
|
||||
],
|
||||
model_group_alias=model_group_alias,
|
||||
)
|
||||
|
||||
|
||||
def test_alias_reports_its_target_groups_limits() -> None:
|
||||
router: Final = _router({"big": "long-context", "item": {"model": "long-context", "hidden": False}})
|
||||
for alias in ("big", "item"):
|
||||
listing = router.get_model_listing_info(alias)
|
||||
assert listing is not None
|
||||
assert (listing.max_input_tokens, listing.max_output_tokens) == (1_000_000, 128_000)
|
||||
row = create_model_info_response(model_id=alias, provider="openai", llm_router=router)
|
||||
assert (row["id"], row.get("max_input_tokens"), row.get("max_output_tokens")) == (alias, 1_000_000, 128_000)
|
||||
|
||||
|
||||
def test_a_deployment_name_wins_over_an_alias_of_the_same_name() -> None:
|
||||
router: Final = _router({"short": "long-context"})
|
||||
listing: Final = router.get_model_listing_info("short")
|
||||
assert listing is not None
|
||||
assert listing.max_input_tokens == 8192
|
||||
|
||||
|
||||
def test_alias_to_an_unknown_group_and_unknown_names_stay_none() -> None:
|
||||
router: Final = _router({"dangling": "no-such-group"})
|
||||
assert router.get_model_listing_info("dangling") is None
|
||||
assert router.get_model_listing_info("not-listed") is None
|
||||
Loading…
Add table
Reference in a new issue