From b41eed4347c169a68de7fc0f9fa25dd0f8bcd450 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 18:22:58 -0700 Subject: [PATCH] fix(proxy): resolve every name a shared PTU deployment is served under to its group's ceiling --- .../hooks/parallel_request_limiter_v3.py | 12 ++-- litellm/router_utils/ptu_shares.py | 52 ++++++++++++----- .../hooks/test_parallel_request_limiter_v3.py | 42 +++++++++++++- tests/unit/router_utils/test_ptu_shares.py | 58 ++++++++++++++----- 4 files changed, 129 insertions(+), 35 deletions(-) diff --git a/litellm/proxy/hooks/parallel_request_limiter_v3.py b/litellm/proxy/hooks/parallel_request_limiter_v3.py index 9fd40115e10..74daca823ac 100644 --- a/litellm/proxy/hooks/parallel_request_limiter_v3.py +++ b/litellm/proxy/hooks/parallel_request_limiter_v3.py @@ -134,7 +134,7 @@ def _resolve_ptu_team_ceiling_via_proxy_router(team_id: str, model_group: str) - if llm_router is None or not is_ptu_cost_attribution_enabled(): return None - return team_ptu_ceiling(llm_router.get_model_list() or (), team_id, model_group) + return team_ptu_ceiling(llm_router.get_model_list() or (), llm_router.model_list, team_id, model_group) FAIL_CLOSED_RATE_LIMIT_ENFORCEMENT_SETTING: Final = "fail_closed_rate_limit_enforcement" @@ -4935,14 +4935,14 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): if reserved_ceiling is not None else self._ptu_team_ceiling_resolver(team_id, reconcile_model.group) ) - scope: Final = ( - PTU_TEAM_DESCRIPTOR_KEY, - f"{team_id}:{ceiling.model_group if ceiling is not None else reconcile_model.group}", + reserved_ptu_scopes: Final = tuple(scope for scope in reserved_scopes if scope[0] == PTU_TEAM_DESCRIPTOR_KEY) + targets: Final = reserved_ptu_scopes or ( + ((PTU_TEAM_DESCRIPTOR_KEY, f"{team_id}:{ceiling.model_group}"),) if ceiling is not None else () ) - if ceiling is None and scope not in reserved_scopes: + if not targets: return () return self._build_reservation_aware_tpm_ops( - targets=(scope,), + targets=targets, reserved_scopes=reserved_scopes, actual_tokens=self._ptu_settlement_tokens( ceiling, self._resolve_reconciled_usage(response_obj), total_tokens diff --git a/litellm/router_utils/ptu_shares.py b/litellm/router_utils/ptu_shares.py index 91ccc585ba7..14ae11dc298 100644 --- a/litellm/router_utils/ptu_shares.py +++ b/litellm/router_utils/ptu_shares.py @@ -57,20 +57,26 @@ def filter_ptu_shared_deployments( def team_ptu_ceiling( - deployments: Sequence[Mapping[str, object]], team_id: str, requested_model: str + listed_rows: Sequence[Mapping[str, object]], + deployments: Sequence[Mapping[str, object]], + team_id: str, + requested_model: str, ) -> PTUTeamCeiling | None: """The per-minute normalized-token ceiling ``team_id``'s shares on the group serving ``requested_model`` add up to, else None when the team holds no share on a deployment with a known sizing row. - A request naming one deployment by its id or provider model counts against that deployment's - group, so every name the router serves it under shares one ceiling. + ``listed_rows`` is every row the router lists a name under, alias and routing-group copies + included, and ``deployments`` is the router's own deployments. A name resolves to the + deployments behind it, so a group, a routing group, a deployment id, and a provider model + all count against the one ceiling of the group whose shared deployment the team can be + served from. Two shared deployments of different models in one group are weighted by the larger output and cached-input ratios, which over-counts those tokens on the cheaper one rather than under-counting them on the dearer one. """ - model_group: Final = _model_group_of(deployments, requested_model) + model_group: Final = _model_group_of(listed_rows, deployments, team_id, requested_model) priced: Final = tuple( (shares[team_id], capacity) for deployment in model_group_deployments(deployments, model_group) @@ -103,10 +109,14 @@ def model_group_deployments(deployments: Sequence[_DeploymentT], model_group: st ) -def _names_deployment(deployment: Mapping[str, object], name: str) -> bool: +def _deployment_id(deployment: Mapping[str, object]) -> object: model_info: Final = deployment.get("model_info") + return model_info.get("id") if isinstance(model_info, Mapping) else None + + +def _names_deployment(deployment: Mapping[str, object], name: str) -> bool: litellm_params: Final = deployment.get("litellm_params") - return (isinstance(model_info, Mapping) and model_info.get("id") == name) or ( + return _deployment_id(deployment) == name or ( isinstance(litellm_params, Mapping) and litellm_params.get("model") == name ) @@ -120,13 +130,29 @@ def _deployment_model_group(deployment: Mapping[str, object]) -> str | None: return model_name if isinstance(model_name, str) else None -def _model_group_of(deployments: Sequence[Mapping[str, object]], requested_model: str) -> str: - """The group ``requested_model`` routes to: itself when it names a group, else the group of - the deployment it names by id or by provider model, the way the router falls back to them.""" - if model_group_deployments(deployments, requested_model): - return requested_model - named: Final = tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model)) - shared_first: Final = sorted(named, key=lambda deployment: _deployment_shares(deployment) is None) +def _routed_deployments( + listed_rows: Sequence[Mapping[str, object]], deployments: Sequence[Mapping[str, object]], requested_model: str +) -> tuple[Mapping[str, object], ...]: + """The router's own deployments behind ``requested_model``: those of the rows listed under it + when it names a group, else the one it names by id or by provider model, the way the router + falls back to them.""" + listed_ids: Final = frozenset(_deployment_id(row) for row in model_group_deployments(listed_rows, requested_model)) + if listed_ids: + return tuple(deployment for deployment in deployments if _deployment_id(deployment) in listed_ids) + return tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model)) + + +def _model_group_of( + listed_rows: Sequence[Mapping[str, object]], + deployments: Sequence[Mapping[str, object]], + team_id: str, + requested_model: str, +) -> str: + """The group of the deployment behind ``requested_model`` this team can be served from, one + holding its share first.""" + routed: Final = _routed_deployments(listed_rows, deployments, requested_model) + servable: Final = filter_ptu_shared_deployments(routed, team_id).deployments + shared_first: Final = sorted(servable, key=lambda deployment: _deployment_shares(deployment) is None) return next( (group for deployment in shared_first if (group := _deployment_model_group(deployment)) is not None), requested_model, diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py index 0b3e54a8358..d22f595b634 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py @@ -7658,7 +7658,8 @@ def _shared_ptu_router(model_group: str) -> Router: "ptu_shares": {"t": 1}, }, } - ] + ], + model_group_alias={f"{model_group}-alias": model_group}, ) @@ -7768,7 +7769,8 @@ async def test_naming_the_shared_deployment_directly_draws_on_the_same_ceiling_a monkeypatch, deployment_name ): """The router also serves a deployment named by its id or its provider model, so a team that - spent its share by group name cannot keep going under the deployment's other names.""" + spent its share by group name cannot keep going under the deployment's other names, even + though the router lists the deployment's alias copy ahead of it.""" monkeypatch.setenv(PTU_COST_ATTRIBUTION_ENV_VAR, "true") cache = DualCache() handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(cache)) @@ -7929,6 +7931,42 @@ async def test_a_reservation_is_settled_even_after_the_teams_share_is_gone(): assert _ptu_increment(handler, ops) == 300 - stash.ptu_reserved_tokens +@pytest.mark.asyncio +async def test_success_settles_the_scope_the_reservation_was_taken_on(): + """Settlement credits the scope admission reserved, not one rebuilt from the name the + request used, so a request naming the deployment by id cannot leave its reservation standing.""" + ceiling: dict[str, PTUTeamCeiling | None] = { + "current": PTUTeamCeiling( + model_group="test-model", tpm_limit=2000, output_to_input_ratio=4.0, cached_input_ratio=0.0 + ) + } + cache = DualCache() + handler = _PROXY_MaxParallelRequestsHandler( + internal_usage_cache=InternalUsageCache(cache), + ptu_team_ceiling_resolver=lambda _team, _group: ceiling["current"], + ) + key = UserAPIKeyAuth(api_key=hash_token("sk-ptu"), team_id="t") + + await handler.async_pre_call_hook( + user_api_key_dict=key, cache=cache, data=_ptu_request("shared-ptu"), call_type="acompletion" + ) + stash = get_request_stash() + assert stash is not None + assert ("model_per_team_ptu", "t:test-model") in stash.reserved_scopes + + ceiling["current"] = None + stash.ptu_ceiling = None + kwargs = _ptu_success_kwargs() + kwargs["litellm_params"]["metadata"]["model_group"] = "shared-ptu" + ops = handler._build_success_event_pipeline_operations( + kwargs=kwargs, + response_obj=_ptu_response(Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)), + rate_limit_type="total", + ) + + assert _ptu_increment(handler, ops) == 150 - stash.ptu_reserved_tokens + + async def _reserve_a_ptu_minute(cache: DualCache, call_id: str) -> tuple[_PROXY_MaxParallelRequestsHandler, str]: resolve, _ = _ptu_ceiling_for("t", "test-model", tpm_limit=2000, ratio=4.0) handler = _PROXY_MaxParallelRequestsHandler( diff --git a/tests/unit/router_utils/test_ptu_shares.py b/tests/unit/router_utils/test_ptu_shares.py index 0012b5628c2..f0cd0cb20bd 100644 --- a/tests/unit/router_utils/test_ptu_shares.py +++ b/tests/unit/router_utils/test_ptu_shares.py @@ -49,6 +49,10 @@ def _single_team(model: str = "azure/gpt-4.1") -> dict: _OPEN: Final = {"model_name": "gpt-4.1-ptu", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "open"}} +def _unaliased_ceiling(deployments: list[dict], team_id: str, requested_model: str) -> PTUTeamCeiling | None: + return team_ptu_ceiling(deployments, deployments, team_id, requested_model) + + def test_a_team_holding_a_share_keeps_the_shared_deployment(): result: Final = filter_ptu_shared_deployments([_shared(), _OPEN], "team-a") assert [d["model_info"]["id"] for d in result.deployments] == ["shared", "open"] @@ -77,7 +81,7 @@ def test_a_single_team_deployment_and_a_malformed_share_map_are_not_filtered_her def test_a_share_converts_to_the_models_input_tpm_per_ptu(): - ceiling: Final = team_ptu_ceiling([_shared()], "team-a", "gpt-4.1-ptu") + ceiling: Final = _unaliased_ceiling([_shared()], "team-a", "gpt-4.1-ptu") assert ceiling == PTUTeamCeiling( model_group="gpt-4.1-ptu", tpm_limit=30 * _GPT41.input_tpm_per_ptu, @@ -88,7 +92,7 @@ def test_a_share_converts_to_the_models_input_tpm_per_ptu(): def test_shares_across_deployments_add_up_and_the_larger_output_ratio_wins(): gpt4o: Final = _shared(model="azure/gpt-4o", shares={"team-a": 10}, deployment_id="shared-4o") - ceiling: Final = team_ptu_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu") + ceiling: Final = _unaliased_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu") assert ceiling is not None assert ceiling.tpm_limit == 30 * _GPT41.input_tpm_per_ptu + 10 * _GPT4O.input_tpm_per_ptu assert ceiling.output_to_input_ratio == max(_GPT41.output_to_input_ratio, _GPT4O.output_to_input_ratio) @@ -98,7 +102,7 @@ def test_the_larger_cached_input_ratio_wins_across_deployments(): """A team sharing two models is weighted by the one that charges more for cache reads, whichever order the deployments come in.""" gpt6sol: Final = _shared(model="azure/gpt-6-sol", shares={"team-a": 10}, deployment_id="shared-6") - ceiling: Final = team_ptu_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu") + ceiling: Final = _unaliased_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu") assert ceiling is not None assert _GPT41.cached_input_ratio < _GPT6SOL.cached_input_ratio assert ceiling.cached_input_ratio == _GPT6SOL.cached_input_ratio @@ -121,9 +125,9 @@ def test_a_group_is_served_by_name_or_by_a_team_scoped_deployments_public_name() def test_no_share_or_no_sizing_row_sets_no_ceiling(): - assert team_ptu_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None - assert team_ptu_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None - assert team_ptu_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None + assert _unaliased_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None + assert _unaliased_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None + assert _unaliased_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_ceiling(): @@ -131,13 +135,13 @@ def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_ name, so those names share the group's ceiling instead of bypassing it.""" payg: Final = {"model_name": "gpt-4.1-payg", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "payg"}} deployments: Final = [payg, _shared(), _OPEN] - by_group: Final = team_ptu_ceiling(deployments, "team-a", "gpt-4.1-ptu") + by_group: Final = _unaliased_ceiling(deployments, "team-a", "gpt-4.1-ptu") assert by_group is not None assert by_group.model_group == "gpt-4.1-ptu" - assert team_ptu_ceiling(deployments, "team-a", "shared") == by_group - assert team_ptu_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group - assert team_ptu_ceiling(deployments, "team-a", "payg") is None - assert team_ptu_ceiling(deployments, "team-a", "missing") is None + assert _unaliased_ceiling(deployments, "team-a", "shared") == by_group + assert _unaliased_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group + assert _unaliased_ceiling(deployments, "team-a", "payg") is None + assert _unaliased_ceiling(deployments, "team-a", "missing") is None def test_a_group_name_wins_over_a_deployment_id_it_collides_with(): @@ -147,7 +151,7 @@ def test_a_group_name_wins_over_a_deployment_id_it_collides_with(): "litellm_params": {"model": "azure/gpt-4o"}, "model_info": {"id": "colliding"}, } - assert team_ptu_ceiling([_shared(), colliding], "team-a", "shared") is None + assert _unaliased_ceiling([_shared(), colliding], "team-a", "shared") is None def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling(): @@ -156,10 +160,36 @@ def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {**_shared()["model_info"], "id": "team-scoped", "team_public_model_name": "gpt-4.1-ptu"}, } - ceiling: Final = team_ptu_ceiling([team_scoped], "team-a", "team-scoped") + ceiling: Final = _unaliased_ceiling([team_scoped], "team-a", "team-scoped") assert ceiling is not None assert ceiling.model_group == "gpt-4.1-ptu" - assert ceiling == team_ptu_ceiling([team_scoped], "team-a", "gpt-4.1-ptu") + assert ceiling == _unaliased_ceiling([team_scoped], "team-a", "gpt-4.1-ptu") + + +def test_alias_and_routing_group_copies_do_not_split_a_deployments_ceiling(): + """The router lists alias and routing-group copies of a deployment under their own names + ahead of its own rows, so every name still resolves to the deployment's group.""" + shared: Final = _shared() + listed: Final = [{**shared, "model_name": "ptu-alias"}, {**shared, "model_name": "ptu-routing-group"}, shared] + by_group: Final = team_ptu_ceiling(listed, [shared], "team-a", "gpt-4.1-ptu") + assert by_group is not None + assert by_group.model_group == "gpt-4.1-ptu" + for name in ("shared", "azure/gpt-4.1", "ptu-routing-group"): + assert team_ptu_ceiling(listed, [shared], "team-a", name) == by_group + + +def test_a_provider_model_draws_on_the_group_where_the_team_holds_its_share(): + """Two groups share deployments of one provider model among different teams, and the router + serves each team only the one it holds a share of.""" + east: Final = _shared(shares={"team-a": 30}, deployment_id="east") + west: Final = {**_shared(shares={"team-b": 20}, deployment_id="west"), "model_name": "gpt-4.1-ptu-west"} + by_provider_model: Final = _unaliased_ceiling([east, west], "team-b", "azure/gpt-4.1") + assert by_provider_model is not None + assert by_provider_model.model_group == "gpt-4.1-ptu-west" + assert by_provider_model == _unaliased_ceiling([east, west], "team-b", "gpt-4.1-ptu-west") + assert _unaliased_ceiling([east, west], "team-a", "azure/gpt-4.1") == _unaliased_ceiling( + [east, west], "team-a", "gpt-4.1-ptu" + ) def test_a_groups_capacity_comes_from_its_first_reserved_deployment_with_a_row():