mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
fix(proxy): resolve every name a shared PTU deployment is served under to its group's ceiling
Some checks are pending
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Terraform Provider / gofmt, vet, build, test (push) Waiting to run
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Waiting to run
Some checks are pending
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Terraform Provider / gofmt, vet, build, test (push) Waiting to run
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Waiting to run
This commit is contained in:
parent
ea8b63d90f
commit
b41eed4347
4 changed files with 129 additions and 35 deletions
|
|
@ -134,7 +134,7 @@ def _resolve_ptu_team_ceiling_via_proxy_router(team_id: str, model_group: str) -
|
|||
|
||||
if llm_router is None or not is_ptu_cost_attribution_enabled():
|
||||
return None
|
||||
return team_ptu_ceiling(llm_router.get_model_list() or (), team_id, model_group)
|
||||
return team_ptu_ceiling(llm_router.get_model_list() or (), llm_router.model_list, team_id, model_group)
|
||||
|
||||
|
||||
FAIL_CLOSED_RATE_LIMIT_ENFORCEMENT_SETTING: Final = "fail_closed_rate_limit_enforcement"
|
||||
|
|
@ -4935,14 +4935,14 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
if reserved_ceiling is not None
|
||||
else self._ptu_team_ceiling_resolver(team_id, reconcile_model.group)
|
||||
)
|
||||
scope: Final = (
|
||||
PTU_TEAM_DESCRIPTOR_KEY,
|
||||
f"{team_id}:{ceiling.model_group if ceiling is not None else reconcile_model.group}",
|
||||
reserved_ptu_scopes: Final = tuple(scope for scope in reserved_scopes if scope[0] == PTU_TEAM_DESCRIPTOR_KEY)
|
||||
targets: Final = reserved_ptu_scopes or (
|
||||
((PTU_TEAM_DESCRIPTOR_KEY, f"{team_id}:{ceiling.model_group}"),) if ceiling is not None else ()
|
||||
)
|
||||
if ceiling is None and scope not in reserved_scopes:
|
||||
if not targets:
|
||||
return ()
|
||||
return self._build_reservation_aware_tpm_ops(
|
||||
targets=(scope,),
|
||||
targets=targets,
|
||||
reserved_scopes=reserved_scopes,
|
||||
actual_tokens=self._ptu_settlement_tokens(
|
||||
ceiling, self._resolve_reconciled_usage(response_obj), total_tokens
|
||||
|
|
|
|||
|
|
@ -57,20 +57,26 @@ def filter_ptu_shared_deployments(
|
|||
|
||||
|
||||
def team_ptu_ceiling(
|
||||
deployments: Sequence[Mapping[str, object]], team_id: str, requested_model: str
|
||||
listed_rows: Sequence[Mapping[str, object]],
|
||||
deployments: Sequence[Mapping[str, object]],
|
||||
team_id: str,
|
||||
requested_model: str,
|
||||
) -> PTUTeamCeiling | None:
|
||||
"""The per-minute normalized-token ceiling ``team_id``'s shares on the group serving
|
||||
``requested_model`` add up to, else None when the team holds no share on a deployment with
|
||||
a known sizing row.
|
||||
|
||||
A request naming one deployment by its id or provider model counts against that deployment's
|
||||
group, so every name the router serves it under shares one ceiling.
|
||||
``listed_rows`` is every row the router lists a name under, alias and routing-group copies
|
||||
included, and ``deployments`` is the router's own deployments. A name resolves to the
|
||||
deployments behind it, so a group, a routing group, a deployment id, and a provider model
|
||||
all count against the one ceiling of the group whose shared deployment the team can be
|
||||
served from.
|
||||
|
||||
Two shared deployments of different models in one group are weighted by the larger
|
||||
output and cached-input ratios, which over-counts those tokens on the cheaper one rather
|
||||
than under-counting them on the dearer one.
|
||||
"""
|
||||
model_group: Final = _model_group_of(deployments, requested_model)
|
||||
model_group: Final = _model_group_of(listed_rows, deployments, team_id, requested_model)
|
||||
priced: Final = tuple(
|
||||
(shares[team_id], capacity)
|
||||
for deployment in model_group_deployments(deployments, model_group)
|
||||
|
|
@ -103,10 +109,14 @@ def model_group_deployments(deployments: Sequence[_DeploymentT], model_group: st
|
|||
)
|
||||
|
||||
|
||||
def _names_deployment(deployment: Mapping[str, object], name: str) -> bool:
|
||||
def _deployment_id(deployment: Mapping[str, object]) -> object:
|
||||
model_info: Final = deployment.get("model_info")
|
||||
return model_info.get("id") if isinstance(model_info, Mapping) else None
|
||||
|
||||
|
||||
def _names_deployment(deployment: Mapping[str, object], name: str) -> bool:
|
||||
litellm_params: Final = deployment.get("litellm_params")
|
||||
return (isinstance(model_info, Mapping) and model_info.get("id") == name) or (
|
||||
return _deployment_id(deployment) == name or (
|
||||
isinstance(litellm_params, Mapping) and litellm_params.get("model") == name
|
||||
)
|
||||
|
||||
|
|
@ -120,13 +130,29 @@ def _deployment_model_group(deployment: Mapping[str, object]) -> str | None:
|
|||
return model_name if isinstance(model_name, str) else None
|
||||
|
||||
|
||||
def _model_group_of(deployments: Sequence[Mapping[str, object]], requested_model: str) -> str:
|
||||
"""The group ``requested_model`` routes to: itself when it names a group, else the group of
|
||||
the deployment it names by id or by provider model, the way the router falls back to them."""
|
||||
if model_group_deployments(deployments, requested_model):
|
||||
return requested_model
|
||||
named: Final = tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model))
|
||||
shared_first: Final = sorted(named, key=lambda deployment: _deployment_shares(deployment) is None)
|
||||
def _routed_deployments(
|
||||
listed_rows: Sequence[Mapping[str, object]], deployments: Sequence[Mapping[str, object]], requested_model: str
|
||||
) -> tuple[Mapping[str, object], ...]:
|
||||
"""The router's own deployments behind ``requested_model``: those of the rows listed under it
|
||||
when it names a group, else the one it names by id or by provider model, the way the router
|
||||
falls back to them."""
|
||||
listed_ids: Final = frozenset(_deployment_id(row) for row in model_group_deployments(listed_rows, requested_model))
|
||||
if listed_ids:
|
||||
return tuple(deployment for deployment in deployments if _deployment_id(deployment) in listed_ids)
|
||||
return tuple(deployment for deployment in deployments if _names_deployment(deployment, requested_model))
|
||||
|
||||
|
||||
def _model_group_of(
|
||||
listed_rows: Sequence[Mapping[str, object]],
|
||||
deployments: Sequence[Mapping[str, object]],
|
||||
team_id: str,
|
||||
requested_model: str,
|
||||
) -> str:
|
||||
"""The group of the deployment behind ``requested_model`` this team can be served from, one
|
||||
holding its share first."""
|
||||
routed: Final = _routed_deployments(listed_rows, deployments, requested_model)
|
||||
servable: Final = filter_ptu_shared_deployments(routed, team_id).deployments
|
||||
shared_first: Final = sorted(servable, key=lambda deployment: _deployment_shares(deployment) is None)
|
||||
return next(
|
||||
(group for deployment in shared_first if (group := _deployment_model_group(deployment)) is not None),
|
||||
requested_model,
|
||||
|
|
|
|||
|
|
@ -7658,7 +7658,8 @@ def _shared_ptu_router(model_group: str) -> Router:
|
|||
"ptu_shares": {"t": 1},
|
||||
},
|
||||
}
|
||||
]
|
||||
],
|
||||
model_group_alias={f"{model_group}-alias": model_group},
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -7768,7 +7769,8 @@ async def test_naming_the_shared_deployment_directly_draws_on_the_same_ceiling_a
|
|||
monkeypatch, deployment_name
|
||||
):
|
||||
"""The router also serves a deployment named by its id or its provider model, so a team that
|
||||
spent its share by group name cannot keep going under the deployment's other names."""
|
||||
spent its share by group name cannot keep going under the deployment's other names, even
|
||||
though the router lists the deployment's alias copy ahead of it."""
|
||||
monkeypatch.setenv(PTU_COST_ATTRIBUTION_ENV_VAR, "true")
|
||||
cache = DualCache()
|
||||
handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(cache))
|
||||
|
|
@ -7929,6 +7931,42 @@ async def test_a_reservation_is_settled_even_after_the_teams_share_is_gone():
|
|||
assert _ptu_increment(handler, ops) == 300 - stash.ptu_reserved_tokens
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_success_settles_the_scope_the_reservation_was_taken_on():
|
||||
"""Settlement credits the scope admission reserved, not one rebuilt from the name the
|
||||
request used, so a request naming the deployment by id cannot leave its reservation standing."""
|
||||
ceiling: dict[str, PTUTeamCeiling | None] = {
|
||||
"current": PTUTeamCeiling(
|
||||
model_group="test-model", tpm_limit=2000, output_to_input_ratio=4.0, cached_input_ratio=0.0
|
||||
)
|
||||
}
|
||||
cache = DualCache()
|
||||
handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(cache),
|
||||
ptu_team_ceiling_resolver=lambda _team, _group: ceiling["current"],
|
||||
)
|
||||
key = UserAPIKeyAuth(api_key=hash_token("sk-ptu"), team_id="t")
|
||||
|
||||
await handler.async_pre_call_hook(
|
||||
user_api_key_dict=key, cache=cache, data=_ptu_request("shared-ptu"), call_type="acompletion"
|
||||
)
|
||||
stash = get_request_stash()
|
||||
assert stash is not None
|
||||
assert ("model_per_team_ptu", "t:test-model") in stash.reserved_scopes
|
||||
|
||||
ceiling["current"] = None
|
||||
stash.ptu_ceiling = None
|
||||
kwargs = _ptu_success_kwargs()
|
||||
kwargs["litellm_params"]["metadata"]["model_group"] = "shared-ptu"
|
||||
ops = handler._build_success_event_pipeline_operations(
|
||||
kwargs=kwargs,
|
||||
response_obj=_ptu_response(Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)),
|
||||
rate_limit_type="total",
|
||||
)
|
||||
|
||||
assert _ptu_increment(handler, ops) == 150 - stash.ptu_reserved_tokens
|
||||
|
||||
|
||||
async def _reserve_a_ptu_minute(cache: DualCache, call_id: str) -> tuple[_PROXY_MaxParallelRequestsHandler, str]:
|
||||
resolve, _ = _ptu_ceiling_for("t", "test-model", tpm_limit=2000, ratio=4.0)
|
||||
handler = _PROXY_MaxParallelRequestsHandler(
|
||||
|
|
|
|||
|
|
@ -49,6 +49,10 @@ def _single_team(model: str = "azure/gpt-4.1") -> dict:
|
|||
_OPEN: Final = {"model_name": "gpt-4.1-ptu", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "open"}}
|
||||
|
||||
|
||||
def _unaliased_ceiling(deployments: list[dict], team_id: str, requested_model: str) -> PTUTeamCeiling | None:
|
||||
return team_ptu_ceiling(deployments, deployments, team_id, requested_model)
|
||||
|
||||
|
||||
def test_a_team_holding_a_share_keeps_the_shared_deployment():
|
||||
result: Final = filter_ptu_shared_deployments([_shared(), _OPEN], "team-a")
|
||||
assert [d["model_info"]["id"] for d in result.deployments] == ["shared", "open"]
|
||||
|
|
@ -77,7 +81,7 @@ def test_a_single_team_deployment_and_a_malformed_share_map_are_not_filtered_her
|
|||
|
||||
|
||||
def test_a_share_converts_to_the_models_input_tpm_per_ptu():
|
||||
ceiling: Final = team_ptu_ceiling([_shared()], "team-a", "gpt-4.1-ptu")
|
||||
ceiling: Final = _unaliased_ceiling([_shared()], "team-a", "gpt-4.1-ptu")
|
||||
assert ceiling == PTUTeamCeiling(
|
||||
model_group="gpt-4.1-ptu",
|
||||
tpm_limit=30 * _GPT41.input_tpm_per_ptu,
|
||||
|
|
@ -88,7 +92,7 @@ def test_a_share_converts_to_the_models_input_tpm_per_ptu():
|
|||
|
||||
def test_shares_across_deployments_add_up_and_the_larger_output_ratio_wins():
|
||||
gpt4o: Final = _shared(model="azure/gpt-4o", shares={"team-a": 10}, deployment_id="shared-4o")
|
||||
ceiling: Final = team_ptu_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu")
|
||||
ceiling: Final = _unaliased_ceiling([_shared(), gpt4o, _OPEN], "team-a", "gpt-4.1-ptu")
|
||||
assert ceiling is not None
|
||||
assert ceiling.tpm_limit == 30 * _GPT41.input_tpm_per_ptu + 10 * _GPT4O.input_tpm_per_ptu
|
||||
assert ceiling.output_to_input_ratio == max(_GPT41.output_to_input_ratio, _GPT4O.output_to_input_ratio)
|
||||
|
|
@ -98,7 +102,7 @@ def test_the_larger_cached_input_ratio_wins_across_deployments():
|
|||
"""A team sharing two models is weighted by the one that charges more for cache reads,
|
||||
whichever order the deployments come in."""
|
||||
gpt6sol: Final = _shared(model="azure/gpt-6-sol", shares={"team-a": 10}, deployment_id="shared-6")
|
||||
ceiling: Final = team_ptu_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu")
|
||||
ceiling: Final = _unaliased_ceiling([_shared(), gpt6sol], "team-a", "gpt-4.1-ptu")
|
||||
assert ceiling is not None
|
||||
assert _GPT41.cached_input_ratio < _GPT6SOL.cached_input_ratio
|
||||
assert ceiling.cached_input_ratio == _GPT6SOL.cached_input_ratio
|
||||
|
|
@ -121,9 +125,9 @@ def test_a_group_is_served_by_name_or_by_a_team_scoped_deployments_public_name()
|
|||
|
||||
|
||||
def test_no_share_or_no_sizing_row_sets_no_ceiling():
|
||||
assert team_ptu_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None
|
||||
assert team_ptu_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None
|
||||
assert team_ptu_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None
|
||||
assert _unaliased_ceiling([_shared()], "team-c", "gpt-4.1-ptu") is None
|
||||
assert _unaliased_ceiling([_shared(model="azure/unknown-deployment")], "team-a", "gpt-4.1-ptu") is None
|
||||
assert _unaliased_ceiling([_single_team(), _OPEN], "team-a", "gpt-4.1-ptu") is None
|
||||
|
||||
|
||||
def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_ceiling():
|
||||
|
|
@ -131,13 +135,13 @@ def test_naming_a_shared_deployment_by_id_or_provider_model_draws_on_its_groups_
|
|||
name, so those names share the group's ceiling instead of bypassing it."""
|
||||
payg: Final = {"model_name": "gpt-4.1-payg", "litellm_params": {"model": "azure/gpt-4.1"}, "model_info": {"id": "payg"}}
|
||||
deployments: Final = [payg, _shared(), _OPEN]
|
||||
by_group: Final = team_ptu_ceiling(deployments, "team-a", "gpt-4.1-ptu")
|
||||
by_group: Final = _unaliased_ceiling(deployments, "team-a", "gpt-4.1-ptu")
|
||||
assert by_group is not None
|
||||
assert by_group.model_group == "gpt-4.1-ptu"
|
||||
assert team_ptu_ceiling(deployments, "team-a", "shared") == by_group
|
||||
assert team_ptu_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group
|
||||
assert team_ptu_ceiling(deployments, "team-a", "payg") is None
|
||||
assert team_ptu_ceiling(deployments, "team-a", "missing") is None
|
||||
assert _unaliased_ceiling(deployments, "team-a", "shared") == by_group
|
||||
assert _unaliased_ceiling(deployments, "team-a", "azure/gpt-4.1") == by_group
|
||||
assert _unaliased_ceiling(deployments, "team-a", "payg") is None
|
||||
assert _unaliased_ceiling(deployments, "team-a", "missing") is None
|
||||
|
||||
|
||||
def test_a_group_name_wins_over_a_deployment_id_it_collides_with():
|
||||
|
|
@ -147,7 +151,7 @@ def test_a_group_name_wins_over_a_deployment_id_it_collides_with():
|
|||
"litellm_params": {"model": "azure/gpt-4o"},
|
||||
"model_info": {"id": "colliding"},
|
||||
}
|
||||
assert team_ptu_ceiling([_shared(), colliding], "team-a", "shared") is None
|
||||
assert _unaliased_ceiling([_shared(), colliding], "team-a", "shared") is None
|
||||
|
||||
|
||||
def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling():
|
||||
|
|
@ -156,10 +160,36 @@ def test_a_team_scoped_deployment_named_by_id_draws_on_its_public_groups_ceiling
|
|||
"litellm_params": {"model": "azure/gpt-4.1"},
|
||||
"model_info": {**_shared()["model_info"], "id": "team-scoped", "team_public_model_name": "gpt-4.1-ptu"},
|
||||
}
|
||||
ceiling: Final = team_ptu_ceiling([team_scoped], "team-a", "team-scoped")
|
||||
ceiling: Final = _unaliased_ceiling([team_scoped], "team-a", "team-scoped")
|
||||
assert ceiling is not None
|
||||
assert ceiling.model_group == "gpt-4.1-ptu"
|
||||
assert ceiling == team_ptu_ceiling([team_scoped], "team-a", "gpt-4.1-ptu")
|
||||
assert ceiling == _unaliased_ceiling([team_scoped], "team-a", "gpt-4.1-ptu")
|
||||
|
||||
|
||||
def test_alias_and_routing_group_copies_do_not_split_a_deployments_ceiling():
|
||||
"""The router lists alias and routing-group copies of a deployment under their own names
|
||||
ahead of its own rows, so every name still resolves to the deployment's group."""
|
||||
shared: Final = _shared()
|
||||
listed: Final = [{**shared, "model_name": "ptu-alias"}, {**shared, "model_name": "ptu-routing-group"}, shared]
|
||||
by_group: Final = team_ptu_ceiling(listed, [shared], "team-a", "gpt-4.1-ptu")
|
||||
assert by_group is not None
|
||||
assert by_group.model_group == "gpt-4.1-ptu"
|
||||
for name in ("shared", "azure/gpt-4.1", "ptu-routing-group"):
|
||||
assert team_ptu_ceiling(listed, [shared], "team-a", name) == by_group
|
||||
|
||||
|
||||
def test_a_provider_model_draws_on_the_group_where_the_team_holds_its_share():
|
||||
"""Two groups share deployments of one provider model among different teams, and the router
|
||||
serves each team only the one it holds a share of."""
|
||||
east: Final = _shared(shares={"team-a": 30}, deployment_id="east")
|
||||
west: Final = {**_shared(shares={"team-b": 20}, deployment_id="west"), "model_name": "gpt-4.1-ptu-west"}
|
||||
by_provider_model: Final = _unaliased_ceiling([east, west], "team-b", "azure/gpt-4.1")
|
||||
assert by_provider_model is not None
|
||||
assert by_provider_model.model_group == "gpt-4.1-ptu-west"
|
||||
assert by_provider_model == _unaliased_ceiling([east, west], "team-b", "gpt-4.1-ptu-west")
|
||||
assert _unaliased_ceiling([east, west], "team-a", "azure/gpt-4.1") == _unaliased_ceiling(
|
||||
[east, west], "team-a", "gpt-4.1-ptu"
|
||||
)
|
||||
|
||||
|
||||
def test_a_groups_capacity_comes_from_its_first_reserved_deployment_with_a_row():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue