fix(proxy): resolve every name a shared PTU deployment is served under to its group's ceiling
Some checks are pending
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Terraform Provider / gofmt, vet, build, test (push) Waiting to run
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Waiting to run

This commit is contained in:
mateo-berri 2026-09-28 18:22:58 -07:00
parent ea8b63d90f
commit b41eed4347
4 changed files with 129 additions and 35 deletions

View file

@ -134,7 +134,7 @@ def _resolve_ptu_team_ceiling_via_proxy_router(team_id: str, model_group: str) -
if llm_router is None or not is_ptu_cost_attribution_enabled():
return None
return team_ptu_ceiling(llm_router.get_model_list() or (), team_id, model_group)
return team_ptu_ceiling(llm_router.get_model_list() or (), llm_router.model_list, team_id, model_group)
FAIL_CLOSED_RATE_LIMIT_ENFORCEMENT_SETTING: Final = "fail_closed_rate_limit_enforcement"
@ -4935,14 +4935,14 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
if reserved_ceiling is not None
else self._ptu_team_ceiling_resolver(team_id, reconcile_model.group)
)
scope: Final = (
PTU_TEAM_DESCRIPTOR_KEY,
f"{team_id}:{ceiling.model_group if ceiling is not None else reconcile_model.group}",
reserved_ptu_scopes: Final = tuple(scope for scope in reserved_scopes if scope[0] == PTU_TEAM_DESCRIPTOR_KEY)
targets: Final = reserved_ptu_scopes or (
((PTU_TEAM_DESCRIPTOR_KEY, f"{team_id}:{ceiling.model_group}"),) if ceiling is not None else ()
)
if ceiling is None and scope not in reserved_scopes:
if not targets:
return ()
return self._build_reservation_aware_tpm_ops(
targets=(scope,),
targets=targets,
reserved_scopes=reserved_scopes,
actual_tokens=self._ptu_settlement_tokens(
ceiling, self._resolve_reconciled_usage(response_obj), total_tokens

View file

@ -57,20 +57,26 @@ def filter_ptu_shared_deployments(
def team_ptu_ceiling(
deployments: Sequence[Mapping[str, object]], team_id: str, requested_model: str
listed_rows: Sequence[Mapping[str, object]],
deployments: Sequence[Mapping[str, object]],
team_id: str,
requested_model: str,
) -> PTUTeamCeiling | None:
"""The per-minute normalized-token ceiling ``team_id``'s shares on the group serving
``requested_model`` add up to, else None when the team holds no share on a deployment with
a known sizing row.
A request naming one deployment by its id or provider model counts against that deployment's
group, so every name the router serves it under shares one ceiling.
``listed_rows`` is every row the router lists a name under, alias and routing-group copies
included, and ``deployments`` is the router's own deployments. A name resolves to the
deployments behind it, so a group, a routing group, a deployment id, and a provider model
all count against the one ceiling of the group whose shared deployment the team can be
served from.
Two shared deployments of different models in one group are weighted by the larger
output and cached-input ratios, which over-counts those tokens on the cheaper one rather
than under-counting them on the dearer one.
"""
model_group: Final = _model_group_of(deployments, requested_model)
model_group: Final = _model_group_of(listed_rows, deployments, team_id, requested_model)
priced: Final = tuple(
(shares[team_id], capacity)
for deployment in model_group_deployments(deployments, model_group)
@ -103,10 +109,14 @@ def model_group_deployments(deployments: Sequence[_DeploymentT], model_group: st
)
def _names_deployment(deployment: Mapping[str, object], name: str) -> bool:
def _deployment_id(deployment: Mapping[str, object]) -> object:
model_info: Final = deployment.get("model_info")
return model_info.get("id") if isinstance(model_info, Mapping) else None
def _names_deployment(deployment: Mapping[str, object], name: str) -> bool:
litellm_params: Final = deployment.get("litellm_params")
return (isinstance(model_info, Mapping) and model_info.get("id") == name) or (
return _deployment_id(deployment) == name or (
isinstance(litellm_params, Mapping) and litellm_params.get("model") == name
)
@ -120,13 +130,29 @@ def _deployment_model_group(deployment: Mapping[str, object]) -> str | None:
return model_name if isinstance(model_name, str) else None
def _model_group_of(deployments: Sequence[Mapping[str, object]], requested_model: str) -> str:
"""The group ``requested_model`` routes to: itself when it names a group, else the group of
the deployment it names by id or by provider model, the way the router falls back to them."""
if model_group_deployments(deployments, requested_model):
return requested_model
named: Final = tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model))
shared_first: Final = sorted(named, key=lambda deployment: _deployment_shares(deployment) is None)
def _routed_deployments(
listed_rows: Sequence[Mapping[str, object]], deployments: Sequence[Mapping[str, object]], requested_model: str
) -> tuple[Mapping[str, object], ...]:
"""The router's own deployments behind ``requested_model``: those of the rows listed under it
when it names a group, else the one it names by id or by provider model, the way the router
falls back to them."""
listed_ids: Final = frozenset(_deployment_id(row) for row in model_group_deployments(listed_rows, requested_model))
if listed_ids:
return tuple(deployment for deployment in deployments if _deployment_id(deployment) in listed_ids)
return tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model))
def _model_group_of(
listed_rows: Sequence[Mapping[str, object]],
deployments: Sequence[Mapping[str, object]],
team_id: str,
requested_model: str,
) -> str:
"""The group of the deployment behind ``requested_model`` this team can be served from, one
holding its share first."""
routed: Final = _routed_deployments(listed_rows, deployments, requested_model)
servable: Final = filter_ptu_shared_deployments(routed, team_id).deployments
shared_first: Final = sorted(servable, key=lambda deployment: _deployment_shares(deployment) is None)
return next(
(group for deployment in shared_first if (group := _deployment_model_group(deployment)) is not None),
requested_model,

View file

@ -7658,7 +7658,8 @@ def _shared_ptu_router(model_group: str) -> Router:
"ptu_shares": {"t": 1},
},
}
]
],
model_group_alias={f"{model_group}-alias": model_group},
)
@ -7768,7 +7769,8 @@ async def test_naming_the_shared_deployment_directly_draws_on_the_same_ceiling_a
monkeypatch, deployment_name
):
"""The router also serves a deployment named by its id or its provider model, so a team that
spent its share by group name cannot keep going under the deployment's other names."""
spent its share by group name cannot keep going under the deployment's other names, even
though the router lists the deployment's alias copy ahead of it."""
monkeypatch.setenv(PTU_COST_ATTRIBUTION_ENV_VAR, "true")
cache = DualCache()
handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(cache))
@ -7929,6 +7931,42 @@ async def test_a_reservation_is_settled_even_after_the_teams_share_is_gone():
assert _ptu_increment(handler, ops) == 300 - stash.ptu_reserved_tokens
@pytest.mark.asyncio
async def test_success_settles_the_scope_the_reservation_was_taken_on():
"""Settlement credits the scope admission reserved, not one rebuilt from the name the
request used, so a request naming the deployment by id cannot leave its reservation standing."""
ceiling: dict[str, PTUTeamCeiling | None] = {
"current": PTUTeamCeiling(
model_group="test-model", tpm_limit=2000, output_to_input_ratio=4.0, cached_input_ratio=0.0
)
}
cache = DualCache()
handler = _PROXY_MaxParallelRequestsHandler(
internal_usage_cache=InternalUsageCache(cache),
ptu_team_ceiling_resolver=lambda _team, _group: ceiling["current"],
)
key = UserAPIKeyAuth(api_key=hash_token("sk-ptu"), team_id="t")
await handler.async_pre_call_hook(
user_api_key_dict=key, cache=cache, data=_ptu_request("shared-ptu"), call_type="acompletion"
)
stash = get_request_stash()
assert stash is not None
assert ("model_per_team_ptu", "t:test-model") in stash.reserved_scopes
ceiling["current"] = None
stash.ptu_ceiling = None
kwargs = _ptu_success_kwargs()
kwargs["litellm_params"]["metadata"]["model_group"] = "shared-ptu"
ops = handler._build_success_event_pipeline_operations(
kwargs=kwargs,
response_obj=_ptu_response(Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)),
rate_limit_type="total",
)
assert _ptu_increment(handler, ops) == 150 - stash.ptu_reserved_tokens
async def _reserve_a_ptu_minute(cache: DualCache, call_id: str) -> tuple[_PROXY_MaxParallelRequestsHandler, str]:
resolve, _ = _ptu_ceiling_for("t", "test-model", tpm_limit=2000, ratio=4.0)
handler = _PROXY_MaxParallelRequestsHandler(

View file

@ -49,6 +49,10 @@ def _single_team(model: str = "azure/gpt-4.1") -> dict:
_OPEN: Final = {"model_name": "gpt-4.1-ptu", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "open"}}
def _unaliased_ceiling(deployments: list[dict], team_id: str, requested_model: str) -> PTUTeamCeiling | None:
return team_ptu_ceiling(deployments, deployments, team_id, requested_model)
def test_a_team_holding_a_share_keeps_the_shared_deployment():
result: Final = filter_ptu_shared_deployments([_shared(), _OPEN], "team-a")
assert [d["model_info"]["id"] for d in result.deployments] == ["shared", "open"]
@ -77,7 +81,7 @@ def test_a_single_team_deployment_and_a_malformed_share_map_are_not_filtered_her
def test_a_share_converts_to_the_models_input_tpm_per_ptu():
ceiling: Final = team_ptu_ceiling([_shared()], "team-a", "gpt-4.1-ptu")
ceiling: Final = _unaliased_ceiling([_shared()], "team-a", "gpt-4.1-ptu")
assert ceiling == PTUTeamCeiling(
model_group="gpt-4.1-ptu",
tpm_limit=30 * _GPT41.input_tpm_per_ptu,
@ -88,7 +92,7 @@ def test_a_share_converts_to_the_models_input_tpm_per_ptu():
def test_shares_across_deployments_add_up_and_the_larger_output_ratio_wins():
gpt4o: Final = _shared(model="azure/gpt-4o", shares={"team-a": 10}, deployment_id="shared-4o")
ceiling: Final = team_ptu_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu")
ceiling: Final = _unaliased_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu")
assert ceiling is not None
assert ceiling.tpm_limit == 30 * _GPT41.input_tpm_per_ptu + 10 * _GPT4O.input_tpm_per_ptu
assert ceiling.output_to_input_ratio == max(_GPT41.output_to_input_ratio, _GPT4O.output_to_input_ratio)
@ -98,7 +102,7 @@ def test_the_larger_cached_input_ratio_wins_across_deployments():
"""A team sharing two models is weighted by the one that charges more for cache reads,
whichever order the deployments come in."""
gpt6sol: Final = _shared(model="azure/gpt-6-sol", shares={"team-a": 10}, deployment_id="shared-6")
ceiling: Final = team_ptu_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu")
ceiling: Final = _unaliased_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu")
assert ceiling is not None
assert _GPT41.cached_input_ratio < _GPT6SOL.cached_input_ratio
assert ceiling.cached_input_ratio == _GPT6SOL.cached_input_ratio
@ -121,9 +125,9 @@ def test_a_group_is_served_by_name_or_by_a_team_scoped_deployments_public_name()
def test_no_share_or_no_sizing_row_sets_no_ceiling():
assert team_ptu_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None
assert team_ptu_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None
assert team_ptu_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None
assert _unaliased_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None
assert _unaliased_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None
assert _unaliased_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None
def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_ceiling():
@ -131,13 +135,13 @@ def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_
name, so those names share the group's ceiling instead of bypassing it."""
payg: Final = {"model_name": "gpt-4.1-payg", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "payg"}}
deployments: Final = [payg, _shared(), _OPEN]
by_group: Final = team_ptu_ceiling(deployments, "team-a", "gpt-4.1-ptu")
by_group: Final = _unaliased_ceiling(deployments, "team-a", "gpt-4.1-ptu")
assert by_group is not None
assert by_group.model_group == "gpt-4.1-ptu"
assert team_ptu_ceiling(deployments, "team-a", "shared") == by_group
assert team_ptu_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group
assert team_ptu_ceiling(deployments, "team-a", "payg") is None
assert team_ptu_ceiling(deployments, "team-a", "missing") is None
assert _unaliased_ceiling(deployments, "team-a", "shared") == by_group
assert _unaliased_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group
assert _unaliased_ceiling(deployments, "team-a", "payg") is None
assert _unaliased_ceiling(deployments, "team-a", "missing") is None
def test_a_group_name_wins_over_a_deployment_id_it_collides_with():
@ -147,7 +151,7 @@ def test_a_group_name_wins_over_a_deployment_id_it_collides_with():
"litellm_params": {"model": "azure/gpt-4o"},
"model_info": {"id": "colliding"},
}
assert team_ptu_ceiling([_shared(), colliding], "team-a", "shared") is None
assert _unaliased_ceiling([_shared(), colliding], "team-a", "shared") is None
def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling():
@ -156,10 +160,36 @@ def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling
"litellm_params": {"model": "azure/gpt-4.1"},
"model_info": {**_shared()["model_info"], "id": "team-scoped", "team_public_model_name": "gpt-4.1-ptu"},
}
ceiling: Final = team_ptu_ceiling([team_scoped], "team-a", "team-scoped")
ceiling: Final = _unaliased_ceiling([team_scoped], "team-a", "team-scoped")
assert ceiling is not None
assert ceiling.model_group == "gpt-4.1-ptu"
assert ceiling == team_ptu_ceiling([team_scoped], "team-a", "gpt-4.1-ptu")
assert ceiling == _unaliased_ceiling([team_scoped], "team-a", "gpt-4.1-ptu")
def test_alias_and_routing_group_copies_do_not_split_a_deployments_ceiling():
"""The router lists alias and routing-group copies of a deployment under their own names
ahead of its own rows, so every name still resolves to the deployment's group."""
shared: Final = _shared()
listed: Final = [{**shared, "model_name": "ptu-alias"}, {**shared, "model_name": "ptu-routing-group"}, shared]
by_group: Final = team_ptu_ceiling(listed, [shared], "team-a", "gpt-4.1-ptu")
assert by_group is not None
assert by_group.model_group == "gpt-4.1-ptu"
for name in ("shared", "azure/gpt-4.1", "ptu-routing-group"):
assert team_ptu_ceiling(listed, [shared], "team-a", name) == by_group
def test_a_provider_model_draws_on_the_group_where_the_team_holds_its_share():
"""Two groups share deployments of one provider model among different teams, and the router
serves each team only the one it holds a share of."""
east: Final = _shared(shares={"team-a": 30}, deployment_id="east")
west: Final = {**_shared(shares={"team-b": 20}, deployment_id="west"), "model_name": "gpt-4.1-ptu-west"}
by_provider_model: Final = _unaliased_ceiling([east, west], "team-b", "azure/gpt-4.1")
assert by_provider_model is not None
assert by_provider_model.model_group == "gpt-4.1-ptu-west"
assert by_provider_model == _unaliased_ceiling([east, west], "team-b", "gpt-4.1-ptu-west")
assert _unaliased_ceiling([east, west], "team-a", "azure/gpt-4.1") == _unaliased_ceiling(
[east, west], "team-a", "gpt-4.1-ptu"
)
def test_a_groups_capacity_comes_from_its_first_reserved_deployment_with_a_row():