mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
perf(router): avoid deep-copying wildcard deployments when building model group info
This commit is contained in:
parent
f699ab6660
commit
c86e435ca5
4 changed files with 116 additions and 31 deletions
|
|
@ -8742,7 +8742,7 @@ class Router:
|
|||
total_itpm: Optional[int] = None
|
||||
total_otpm: Optional[int] = None
|
||||
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None
|
||||
model_list = self.get_model_list(model_name=model_group)
|
||||
model_list = self.get_model_list(model_name=model_group, readonly_wildcard_deployments=True)
|
||||
if model_list is None:
|
||||
return None
|
||||
is_wildcard_group = self.pattern_router.is_match(model_group)
|
||||
|
|
@ -9638,12 +9638,19 @@ class Router:
|
|||
return returned_models
|
||||
|
||||
def get_model_list(
|
||||
self, model_name: Optional[str] = None, team_id: Optional[str] = None
|
||||
self,
|
||||
model_name: Optional[str] = None,
|
||||
team_id: Optional[str] = None,
|
||||
readonly_wildcard_deployments: bool = False,
|
||||
) -> Optional[List[DeploymentTypedDict]]:
|
||||
"""
|
||||
Includes router model_group_alias'es as well
|
||||
|
||||
if team_id specified, returns matching team-specific models
|
||||
|
||||
readonly_wildcard_deployments: when True, wildcard matches share nested deployment
|
||||
structures (e.g. model_info) by reference instead of being deep-copied. Only pass True
|
||||
from read-only callers that never mutate the returned deployments.
|
||||
"""
|
||||
# Note: model_list and model_group_alias are always initialized in __init__
|
||||
# so hasattr checks are unnecessary
|
||||
|
|
@ -9655,11 +9662,16 @@ class Router:
|
|||
returned_models.extend(self.get_model_list_from_model_alias(model_name=model_name))
|
||||
|
||||
if len(returned_models) == 0: # check if wildcard route
|
||||
potential_wildcard_models = self.pattern_router.route(model_name) or []
|
||||
route_wildcard = (
|
||||
self.pattern_router.route_readonly if readonly_wildcard_deployments else self.pattern_router.route
|
||||
)
|
||||
potential_wildcard_models = route_wildcard(model_name) or []
|
||||
|
||||
## check for team-specific wildcard models
|
||||
if team_id is not None and team_id in self.team_pattern_routers:
|
||||
potential_team_only_wildcard_models = self.team_pattern_routers[team_id].route(model_name) or []
|
||||
team_router = self.team_pattern_routers[team_id]
|
||||
route_team_wildcard = team_router.route_readonly if readonly_wildcard_deployments else team_router.route
|
||||
potential_team_only_wildcard_models = route_team_wildcard(model_name) or []
|
||||
potential_wildcard_models.extend(potential_team_only_wildcard_models)
|
||||
|
||||
if model_name is not None and potential_wildcard_models is not None:
|
||||
|
|
|
|||
|
|
@ -119,19 +119,51 @@ class PatternMatchRouter:
|
|||
"""
|
||||
if request is None:
|
||||
return None
|
||||
|
||||
sorted_patterns = PatternUtils.sorted_patterns(self.patterns)
|
||||
regex_filtered_model_names = (
|
||||
[self._pattern_to_regex(m) for m in filtered_model_names] if filtered_model_names is not None else []
|
||||
)
|
||||
for pattern, llm_deployments in sorted_patterns:
|
||||
if filtered_model_names is not None and pattern not in regex_filtered_model_names:
|
||||
continue
|
||||
pattern_match = re.match(pattern, request)
|
||||
if pattern_match:
|
||||
return pattern_match, llm_deployments
|
||||
try:
|
||||
sorted_patterns = PatternUtils.sorted_patterns(self.patterns)
|
||||
regex_filtered_model_names = (
|
||||
[self._pattern_to_regex(m) for m in filtered_model_names] if filtered_model_names is not None else []
|
||||
)
|
||||
for pattern, llm_deployments in sorted_patterns:
|
||||
if filtered_model_names is not None and pattern not in regex_filtered_model_names:
|
||||
continue
|
||||
pattern_match = re.match(pattern, request)
|
||||
if pattern_match:
|
||||
return pattern_match, llm_deployments
|
||||
except Exception as e:
|
||||
verbose_router_logger.debug(f"Error in PatternMatchRouter._find_matching_pattern: {str(e)}")
|
||||
return None
|
||||
|
||||
def _shallow_renamed_deployments(self, matched_pattern: Match, deployments: list[dict]) -> list[dict]:
|
||||
return [
|
||||
{
|
||||
**deployment,
|
||||
"litellm_params": {
|
||||
**deployment["litellm_params"],
|
||||
"model": PatternMatchRouter.set_deployment_model_name(
|
||||
matched_pattern=matched_pattern,
|
||||
litellm_deployment_litellm_model=deployment["litellm_params"]["model"],
|
||||
),
|
||||
},
|
||||
}
|
||||
for deployment in deployments
|
||||
]
|
||||
|
||||
def route_readonly(self, request: str | None, filtered_model_names: list[str] | None = None) -> list[dict] | None:
|
||||
"""
|
||||
Like ``route``, but returns shallow copies that share nested deployment structures
|
||||
(notably the potentially large ``model_info``) by reference instead of deep-copying them.
|
||||
|
||||
Only for read-only consumers that never mutate the returned deployments. Deep-copying
|
||||
every matched deployment for every model group is what pegged CPU on GET /v1/models
|
||||
for wildcard routes (see #33636).
|
||||
"""
|
||||
matched = self._find_matching_pattern(request, filtered_model_names)
|
||||
if matched is None:
|
||||
return None
|
||||
pattern_match, llm_deployments = matched
|
||||
return self._shallow_renamed_deployments(pattern_match, llm_deployments)
|
||||
|
||||
def is_match(self, request: str | None, filtered_model_names: list[str] | None = None) -> bool:
|
||||
"""
|
||||
Return whether request matches a wildcard pattern, without deep-copying deployments.
|
||||
|
|
@ -139,11 +171,7 @@ class PatternMatchRouter:
|
|||
Use this instead of ``route(...) is not None`` when only the boolean result is needed;
|
||||
``route`` deep-copies every matched deployment, which is expensive on hot paths.
|
||||
"""
|
||||
try:
|
||||
return self._find_matching_pattern(request, filtered_model_names) is not None
|
||||
except Exception as e:
|
||||
verbose_router_logger.debug(f"Error in PatternMatchRouter.is_match: {str(e)}")
|
||||
return False
|
||||
return self._find_matching_pattern(request, filtered_model_names) is not None
|
||||
|
||||
def route(self, request: Optional[str], filtered_model_names: Optional[List[str]] = None) -> Optional[List[Dict]]:
|
||||
"""
|
||||
|
|
@ -159,16 +187,11 @@ class PatternMatchRouter:
|
|||
Returns:
|
||||
Optional[List[Deployment]]: llm deployments
|
||||
"""
|
||||
try:
|
||||
matched = self._find_matching_pattern(request, filtered_model_names)
|
||||
if matched is None:
|
||||
return None
|
||||
pattern_match, llm_deployments = matched
|
||||
return self._return_pattern_matched_deployments(matched_pattern=pattern_match, deployments=llm_deployments)
|
||||
except Exception as e:
|
||||
verbose_router_logger.debug(f"Error in PatternMatchRouter.route: {str(e)}")
|
||||
|
||||
return None # No matching pattern found
|
||||
matched = self._find_matching_pattern(request, filtered_model_names)
|
||||
if matched is None:
|
||||
return None
|
||||
pattern_match, llm_deployments = matched
|
||||
return self._return_pattern_matched_deployments(matched_pattern=pattern_match, deployments=llm_deployments)
|
||||
|
||||
@staticmethod
|
||||
def set_deployment_model_name(
|
||||
|
|
|
|||
|
|
@ -13,6 +13,15 @@ def _openai_wildcard_router() -> PatternMatchRouter:
|
|||
return router
|
||||
|
||||
|
||||
def _openai_wildcard_router_with_model_info(model_info: dict) -> PatternMatchRouter:
|
||||
router = PatternMatchRouter()
|
||||
router.add_pattern(
|
||||
"openai/*",
|
||||
{"model_name": "openai/*", "litellm_params": {"model": "openai/*"}, "model_info": model_info},
|
||||
)
|
||||
return router
|
||||
|
||||
|
||||
def test_is_match_reports_membership_without_deepcopy():
|
||||
"""
|
||||
Regression for #33636: is_match must report pattern membership without the
|
||||
|
|
@ -48,3 +57,39 @@ def test_route_still_copies_matched_deployments():
|
|||
assert deployments is not None
|
||||
assert deployments[0]["litellm_params"]["model"] == "openai/gpt-4o-mini"
|
||||
assert spy_deepcopy.call_count >= 1
|
||||
|
||||
|
||||
def test_route_readonly_renames_without_deepcopy():
|
||||
"""
|
||||
Regression for #33636: route_readonly rewrites the deployment model to the concrete
|
||||
requested name (same as route) but must not deep-copy the deployment, which is what
|
||||
pegged CPU on GET /v1/models for wildcard routes.
|
||||
"""
|
||||
router = _openai_wildcard_router()
|
||||
|
||||
with patch.object(copy_module, "deepcopy", wraps=copy_module.deepcopy) as spy_deepcopy:
|
||||
deployments = router.route_readonly("openai/gpt-4o-mini")
|
||||
|
||||
assert deployments is not None
|
||||
assert deployments[0]["litellm_params"]["model"] == "openai/gpt-4o-mini"
|
||||
assert spy_deepcopy.call_count == 0
|
||||
|
||||
|
||||
def test_route_readonly_shares_model_info_but_isolates_model_rename():
|
||||
model_info = {"max_input_tokens": 128000, "nested": {"a": 1}}
|
||||
router = _openai_wildcard_router_with_model_info(model_info)
|
||||
stored = router.patterns["openai/(.*)"][0]
|
||||
|
||||
deployments = router.route_readonly("openai/gpt-4o-mini")
|
||||
assert deployments is not None
|
||||
result = deployments[0]
|
||||
|
||||
assert result["model_info"] is model_info
|
||||
assert result["litellm_params"]["model"] == "openai/gpt-4o-mini"
|
||||
assert stored["litellm_params"]["model"] == "openai/*"
|
||||
|
||||
|
||||
def test_route_readonly_returns_none_for_no_match():
|
||||
router = _openai_wildcard_router()
|
||||
assert router.route_readonly("anthropic/claude-3") is None
|
||||
assert router.route_readonly(None) is None
|
||||
|
|
|
|||
|
|
@ -107,6 +107,10 @@ def test_set_model_group_info_wildcard_avoids_per_deployment_deepcopy():
|
|||
"route",
|
||||
wraps=router.pattern_router.route,
|
||||
) as spy_route, patch.object(
|
||||
router.pattern_router,
|
||||
"route_readonly",
|
||||
wraps=router.pattern_router.route_readonly,
|
||||
) as spy_route_readonly, patch.object(
|
||||
router.pattern_router,
|
||||
"is_match",
|
||||
wraps=router.pattern_router.is_match,
|
||||
|
|
@ -114,7 +118,8 @@ def test_set_model_group_info_wildcard_avoids_per_deployment_deepcopy():
|
|||
info = router.get_model_group_info("openai/gpt-4o-mini")
|
||||
|
||||
assert info is not None
|
||||
assert spy_route.call_count <= 1
|
||||
assert spy_route.call_count == 0
|
||||
assert spy_route_readonly.call_count >= 1
|
||||
assert spy_is_match.call_count >= 1
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue