From 1480b6b4888501f88f327d5d3e0ad7bde1c3cc15 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Tue, 22 Sep 2026 15:37:03 +0200 Subject: [PATCH] fix(realtime): count concurrency limits as Live managed constraints `_managed_constraints()` listed every rpm, tpm, and budget limit a key can carry and skipped `max_parallel_requests`, so a key with a configured concurrency limit could still choose managed Responses delegation. Those delegated invocations run upstream and never pass limiter admission, so the key ran concurrent backend calls past the cap it advertises. An admin-configured `global_max_parallel_requests` left the same hole for every key. Managed delegation now rejects either parallel limit being active, the way it already rejects the key's rpm, tpm, and budget limits, and the way `_live_budget_configured` already treats a budget table's `max_parallel_requests`. The tests that assert delegation stays unmanaged pin `general_settings`, so they no longer depend on what an earlier test left there. Validation: 550 passed across `tests/test_litellm/proxy/realtime_endpoints` and `tests/unit/realtime_api`, up from 546 with the four new cases; `basedpyright` reports 0 errors on `live.py`; `ruff check`, `ruff format --check`, and `scripts/test_quality_gate.py --base HEAD` clean. --- litellm/proxy/realtime_endpoints/live.py | 14 ++++++ .../proxy/realtime_endpoints/test_live.py | 45 +++++++++++++++++-- 2 files changed, 56 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/realtime_endpoints/live.py b/litellm/proxy/realtime_endpoints/live.py index 7d75c2100fa..cc1f932a1b7 100644 --- a/litellm/proxy/realtime_endpoints/live.py +++ b/litellm/proxy/realtime_endpoints/live.py @@ -524,6 +524,8 @@ def _session_policy(body: Mapping[str, JsonValue], source: LiveHandle | None) -> def _managed_constraints(auth: UserAPIKeyAuth) -> bool: + from litellm.proxy import proxy_server as server + if any( value is not None for value in ( @@ -539,6 +541,7 @@ def _managed_constraints(auth: UserAPIKeyAuth) -> bool: auth.team_member_tpm_limit, auth.end_user_rpm_limit, auth.end_user_tpm_limit, + auth.max_parallel_requests, auth.max_budget, auth.team_max_budget, auth.user_max_budget, @@ -548,6 +551,17 @@ def _managed_constraints(auth: UserAPIKeyAuth) -> bool: ): return True + # An admin-configured proxy-wide concurrency cap admits every key through the limiter, + # and delegated backend invocations never reach that admission, so it constrains the key + # the same way a key-level `max_parallel_requests` does. + if ( + _MAPPING.validate_python(getattr(server, "general_settings", None) or _EMPTY).get( + "global_max_parallel_requests" + ) + is not None + ): + return True + direct_maps: Final[tuple[object, ...]] = ( _OBJECT_VALUE.validate_python(getattr(auth, "model_max_budget", None)), _OBJECT_VALUE.validate_python(getattr(auth, "user_model_max_budget", None)), diff --git a/tests/test_litellm/proxy/realtime_endpoints/test_live.py b/tests/test_litellm/proxy/realtime_endpoints/test_live.py index 950dd384054..a395735ad12 100644 --- a/tests/test_litellm/proxy/realtime_endpoints/test_live.py +++ b/tests/test_litellm/proxy/realtime_endpoints/test_live.py @@ -465,6 +465,36 @@ def test_managed_constraints_detect_scalar_and_window_budgets(limits): assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", **limits)) is True +@pytest.mark.parametrize("global_limit", [None, 8]) +def test_managed_constraints_covers_configured_parallel_limits(monkeypatch, global_limit): + from litellm.proxy import proxy_server + + monkeypatch.setattr(proxy_server, "general_settings", {"global_max_parallel_requests": global_limit}) + + key_limited = live._managed_constraints(UserAPIKeyAuth(api_key="owner", max_parallel_requests=2)) + globally_limited = live._managed_constraints(UserAPIKeyAuth(api_key="owner")) + assert key_limited is True + assert globally_limited is (global_limit is not None) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("limits", [{}, {"max_parallel_requests": 2}]) +async def test_managed_delegation_requires_client_delegation_under_a_concurrency_limit(monkeypatch, limits): + from litellm.proxy import proxy_server + + monkeypatch.setattr(proxy_server, "general_settings", {}) + body = {"session": {"delegation": {"type": "responses", "responses": {"model": "backend"}}}} + auth = UserAPIKeyAuth(api_key="owner", **limits) + + if not limits: + assert await live._authorize_delegation(body, auth) is None + return + with pytest.raises(HTTPException) as rejected: + await live._authorize_delegation(body, auth) + assert rejected.value.status_code == 400 + assert "use client delegation" in str(rejected.value.detail) + + @pytest.mark.asyncio @pytest.mark.parametrize( "member_limit, default_limit, blocked", @@ -540,11 +570,17 @@ def test_managed_constraints_fails_closed_after_metadata_node_limit(): {"metadata": {"nested": [{"model_max_budget": {}}]}}, ], ) -def test_empty_model_limit_maps_do_not_mark_delegation_as_managed(limits): +def test_empty_model_limit_maps_do_not_mark_delegation_as_managed(monkeypatch, limits): + from litellm.proxy import proxy_server + + monkeypatch.setattr(proxy_server, "general_settings", {}) assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", **limits)) is False -def test_managed_constraints_terminates_on_cyclic_metadata_without_a_limit(): +def test_managed_constraints_terminates_on_cyclic_metadata_without_a_limit(monkeypatch): + from litellm.proxy import proxy_server + + monkeypatch.setattr(proxy_server, "general_settings", {}) metadata = {} metadata["self"] = metadata auth = UserAPIKeyAuth(api_key="owner") @@ -1247,7 +1283,10 @@ async def test_sparse_responses_update_without_model_remains_valid_for_user_scop assert result is None -def test_managed_constraints_uses_exact_metadata_keys(): +def test_managed_constraints_uses_exact_metadata_keys(monkeypatch): + from litellm.proxy import proxy_server + + monkeypatch.setattr(proxy_server, "general_settings", {}) for key in ("max_budget_alert_emails", "model_max_budget_usage"): assert live._managed_constraints(UserAPIKeyAuth(api_key="owner", metadata={key: {"backend": 1}})) is False