fix(realtime): count concurrency limits as Live managed constraints

`_managed_constraints()` listed every rpm, tpm, and budget limit a key can
carry and skipped `max_parallel_requests`, so a key with a configured
concurrency limit could still choose managed Responses delegation. Those
delegated invocations run upstream and never pass limiter admission, so the key
ran concurrent backend calls past the cap it advertises. An admin-configured
`global_max_parallel_requests` left the same hole for every key.

Managed delegation now rejects either parallel limit being active, the way it
already rejects the key's rpm, tpm, and budget limits, and the way
`_live_budget_configured` already treats a budget table's
`max_parallel_requests`. The tests that assert delegation stays unmanaged pin
`general_settings`, so they no longer depend on what an earlier test left there.

Validation: 550 passed across `tests/test_litellm/proxy/realtime_endpoints` and
`tests/unit/realtime_api`, up from 546 with the four new cases;
`basedpyright` reports 0 errors on `live.py`; `ruff check`, `ruff format
--check`, and `scripts/test_quality_gate.py --base HEAD` clean.
This commit is contained in:
jibanez-staticduo 2026-09-22 15:37:03 +02:00
parent 35c00b7093
commit 1480b6b488
No known key found for this signature in database
2 changed files with 56 additions and 3 deletions

View file

@ -524,6 +524,8 @@ def _session_policy(body: Mapping[str, JsonValue], source: LiveHandle | None) ->
def _managed_constraints(auth: UserAPIKeyAuth) -> bool:
from litellm.proxy import proxy_server as server
if any(
value is not None
for value in (
@ -539,6 +541,7 @@ def _managed_constraints(auth: UserAPIKeyAuth) -> bool:
auth.team_member_tpm_limit,
auth.end_user_rpm_limit,
auth.end_user_tpm_limit,
auth.max_parallel_requests,
auth.max_budget,
auth.team_max_budget,
auth.user_max_budget,
@ -548,6 +551,17 @@ def _managed_constraints(auth: UserAPIKeyAuth) -> bool:
):
return True
# An admin-configured proxy-wide concurrency cap admits every key through the limiter,
# and delegated backend invocations never reach that admission, so it constrains the key
# the same way a key-level `max_parallel_requests` does.
if (
_MAPPING.validate_python(getattr(server, "general_settings", None) or _EMPTY).get(
"global_max_parallel_requests"
)
is not None
):
return True
direct_maps: Final[tuple[object, ...]] = (
_OBJECT_VALUE.validate_python(getattr(auth, "model_max_budget", None)),
_OBJECT_VALUE.validate_python(getattr(auth, "user_model_max_budget", None)),

View file

@ -465,6 +465,36 @@ def test_managed_constraints_detect_scalar_and_window_budgets(limits):
assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", **limits)) is True
@pytest.mark.parametrize("global_limit", [None, 8])
def test_managed_constraints_covers_configured_parallel_limits(monkeypatch, global_limit):
from litellm.proxy import proxy_server
monkeypatch.setattr(proxy_server, "general_settings", {"global_max_parallel_requests": global_limit})
key_limited = live._managed_constraints(UserAPIKeyAuth(api_key="owner", max_parallel_requests=2))
globally_limited = live._managed_constraints(UserAPIKeyAuth(api_key="owner"))
assert key_limited is True
assert globally_limited is (global_limit is not None)
@pytest.mark.asyncio
@pytest.mark.parametrize("limits", [{}, {"max_parallel_requests": 2}])
async def test_managed_delegation_requires_client_delegation_under_a_concurrency_limit(monkeypatch, limits):
from litellm.proxy import proxy_server
monkeypatch.setattr(proxy_server, "general_settings", {})
body = {"session": {"delegation": {"type": "responses", "responses": {"model": "backend"}}}}
auth = UserAPIKeyAuth(api_key="owner", **limits)
if not limits:
assert await live._authorize_delegation(body, auth) is None
return
with pytest.raises(HTTPException) as rejected:
await live._authorize_delegation(body, auth)
assert rejected.value.status_code == 400
assert "use client delegation" in str(rejected.value.detail)
@pytest.mark.asyncio
@pytest.mark.parametrize(
"member_limit, default_limit, blocked",
@ -540,11 +570,17 @@ def test_managed_constraints_fails_closed_after_metadata_node_limit():
{"metadata": {"nested": [{"model_max_budget": {}}]}},
],
)
def test_empty_model_limit_maps_do_not_mark_delegation_as_managed(limits):
def test_empty_model_limit_maps_do_not_mark_delegation_as_managed(monkeypatch, limits):
from litellm.proxy import proxy_server
monkeypatch.setattr(proxy_server, "general_settings", {})
assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", **limits)) is False
def test_managed_constraints_terminates_on_cyclic_metadata_without_a_limit():
def test_managed_constraints_terminates_on_cyclic_metadata_without_a_limit(monkeypatch):
from litellm.proxy import proxy_server
monkeypatch.setattr(proxy_server, "general_settings", {})
metadata = {}
metadata["self"] = metadata
auth = UserAPIKeyAuth(api_key="owner")
@ -1247,7 +1283,10 @@ async def test_sparse_responses_update_without_model_remains_valid_for_user_scop
assert result is None
def test_managed_constraints_uses_exact_metadata_keys():
def test_managed_constraints_uses_exact_metadata_keys(monkeypatch):
from litellm.proxy import proxy_server
monkeypatch.setattr(proxy_server, "general_settings", {})
for key in ("max_budget_alert_emails", "model_max_budget_usage"):
assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", metadata={key: {"backend": 1}})) is False