From 4a5828ed7a8238617e0c44289d76814fb68beaca Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 2 Oct 2026 15:29:19 -0700 Subject: [PATCH] fix(ptu): size usage rows by the name the request used --- .../management_endpoints/ptu_consumption.py | 32 +++++++++++---- litellm/router_utils/ptu_shares.py | 4 +- .../test_ptu_consumption.py | 41 ++++++++++++++++++- 3 files changed, 65 insertions(+), 12 deletions(-) diff --git a/litellm/proxy/management_endpoints/ptu_consumption.py b/litellm/proxy/management_endpoints/ptu_consumption.py index 2c8d7d1930e..7d13c293d70 100644 --- a/litellm/proxy/management_endpoints/ptu_consumption.py +++ b/litellm/proxy/management_endpoints/ptu_consumption.py @@ -12,7 +12,8 @@ from typing import Final from litellm.litellm_core_utils.ptu_pricing import is_ptu_cost_attribution_enabled from litellm.llms.azure.ptu_capacity import PTUCapacity, normalized_tokens, ptu_hours from litellm.router import Router -from litellm.router_utils.ptu_shares import model_group_deployments, model_group_ptu_capacity +from litellm.router_utils.common_utils import resolve_model_group_alias +from litellm.router_utils.ptu_shares import model_group_ptu_capacity, routed_deployments from litellm.types.proxy.management_endpoints.common_daily_activity import ( DailySpendData, MetricWithMetadata, @@ -98,16 +99,31 @@ def attach_ptu_hours( ) +def capacity_by_requested_name(llm_router: Router) -> Callable[[str], PTUCapacity | None]: + """The sizing row behind the name a usage row is keyed by, resolved the way the ceiling + resolves a request: an alias to its group, then a group, a routing group, a deployment id, + or a provider model to the deployments it is served from.""" + listed_rows: Final = llm_router.get_model_list() or () + aliases: Final = llm_router.model_group_alias + + def capacity_for(requested_model: str) -> PTUCapacity | None: + model_group: Final = resolve_model_group_alias(aliases, requested_model) or requested_model + return model_group_ptu_capacity( + routed_deployments( + listed_rows, + llm_router.model_list, # pyright: ignore[reportUnknownArgumentType] # Router.model_list is a bare list + model_group, + ) + ) + + return capacity_for + + def with_ptu_consumption( activity: SpendAnalyticsPaginatedResponse, llm_router: Router | None ) -> SpendAnalyticsPaginatedResponse: - """``activity`` with PTU-hours attached from the router's sized model groups, untouched while + """``activity`` with PTU-hours attached from the router's sized deployments, untouched while PTU cost attribution is off or no router is loaded.""" if llm_router is None or not is_ptu_cost_attribution_enabled(): return activity - return attach_ptu_hours( - activity, - lambda model_group: model_group_ptu_capacity( - model_group_deployments(llm_router.get_model_list() or (), model_group) - ), - ) + return attach_ptu_hours(activity, capacity_by_requested_name(llm_router)) diff --git a/litellm/router_utils/ptu_shares.py b/litellm/router_utils/ptu_shares.py index d4f5ffc6e18..ed59a3a40d1 100644 --- a/litellm/router_utils/ptu_shares.py +++ b/litellm/router_utils/ptu_shares.py @@ -130,7 +130,7 @@ def _deployment_model_group(deployment: Mapping[str, object]) -> str | None: return model_name if isinstance(model_name, str) else None -def _routed_deployments( +def routed_deployments( listed_rows: Sequence[Mapping[str, object]], deployments: Sequence[Mapping[str, object]], requested_model: str ) -> tuple[Mapping[str, object], ...]: """The router's own deployments behind ``requested_model``: those of the rows listed under it @@ -150,7 +150,7 @@ def _model_group_of( ) -> str: """The group of the deployment behind ``requested_model`` this team can be served from, one holding its share first.""" - routed: Final = _routed_deployments(listed_rows, deployments, requested_model) + routed: Final = routed_deployments(listed_rows, deployments, requested_model) servable: Final = filter_ptu_shared_deployments(routed, team_id).deployments shared_first: Final = sorted(servable, key=lambda deployment: _deployment_shares(deployment) is None) return next( diff --git a/tests/unit/proxy/management_endpoints/test_ptu_consumption.py b/tests/unit/proxy/management_endpoints/test_ptu_consumption.py index 432bf858899..0b4b048aa1b 100644 --- a/tests/unit/proxy/management_endpoints/test_ptu_consumption.py +++ b/tests/unit/proxy/management_endpoints/test_ptu_consumption.py @@ -4,8 +4,9 @@ from typing import Final import pytest -from litellm.llms.azure.ptu_capacity import PTUCapacity -from litellm.proxy.management_endpoints.ptu_consumption import attach_ptu_hours +from litellm import Router +from litellm.llms.azure.ptu_capacity import AZURE_PTU_CAPACITY, PTUCapacity +from litellm.proxy.management_endpoints.ptu_consumption import attach_ptu_hours, with_ptu_consumption from litellm.types.proxy.management_endpoints.common_daily_activity import ( BreakdownMetrics, DailySpendData, @@ -19,6 +20,7 @@ from litellm.types.proxy.management_endpoints.common_daily_activity import ( _ROW: Final = PTUCapacity(input_tpm_per_ptu=1_000, output_to_input_ratio=4.0) _CACHED_ROW: Final = PTUCapacity(input_tpm_per_ptu=1_000, output_to_input_ratio=4.0, cached_input_ratio=0.1) _CAPACITY: Final = {"gpt-4.1-ptu": _ROW, "gpt-6-ptu": _CACHED_ROW} +_ONE_PTU_HOUR_OF_INPUT: Final = AZURE_PTU_CAPACITY["gpt-4.1"].normalized_tokens_per_ptu_hour def _metrics(prompt: int, completion: int, cached: int = 0) -> SpendMetrics: @@ -128,3 +130,38 @@ def test_a_page_with_no_ptu_group_is_returned_unchanged(): assert attached.results[0] is day assert attached.metadata.total_ptu_hours == 0.0 assert response.metadata.total_ptu_hours == 0.0 + + +def _shared_ptu_router() -> Router: + return Router( + model_list=[ + { + "model_name": "gpt-4.1-ptu", + "litellm_params": {"model": "azure/gpt-4.1", "api_key": "sk-ptu", "api_base": "https://ptu.example"}, + "model_info": { + "id": "shared-ptu", + "base_model": "azure/gpt-4.1", + "ptu_count": 50, + "cost_per_ptu_per_hour": 1.0, + "ptu_effective_from": "2026-01-01T00:00:00Z", + "ptu_shares": {"team-a": 30, "team-b": 20}, + }, + } + ], + model_group_alias={"ptu-alias": {"model": "gpt-4.1-ptu", "hidden": True}}, + ) + + +def test_rows_keyed_by_an_alias_a_deployment_id_or_a_provider_model_are_sized_like_the_group(monkeypatch): + """The ceiling charges a request however it names the shared deployment, so the usage row + that request lands in, keyed by the name it used, reports the same PTU-hours as the group.""" + monkeypatch.setenv("LITELLM_ENABLE_PTU_COST_ATTRIBUTION", "True") + names: Final = ("gpt-4.1-ptu", "ptu-alias", "shared-ptu", "azure/gpt-4.1") + one_hour_each: Final = {name: _bucket(_metrics(prompt=_ONE_PTU_HOUR_OF_INPUT, completion=0)) for name in names} + + attached: Final = with_ptu_consumption(_response(_day("2026-09-23", one_hour_each)), _shared_ptu_router()) + + groups: Final = attached.results[0].breakdown.model_groups + assert [groups[name].metrics.ptu_hours for name in names] == [pytest.approx(1.0)] * len(names) + assert attached.results[0].metrics.ptu_hours == pytest.approx(float(len(names))) + assert attached.metadata.total_ptu_hours == pytest.approx(float(len(names)))