From 91e60b80b095068303c121bff4ceba28742dad41 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 23 Jun 2026 17:53:26 +0530 Subject: [PATCH] fix(key-management): cap CLI session token delegation budget to team ceiling A CLI session token intentionally carries max_budget=None to avoid a per-session LLM spend cap. The key-generation delegation check (GHSA-q775-qw9r-2r4g) previously skipped non-admin callers with max_budget=None, treating them as having unlimited delegation authority. This allowed any internal user with a lite login session to mint virtual keys with arbitrary budgets. Adds is_session_token=True to UserAPIKeyAuth for CLI session tokens and uses the caller's team budget as the delegation ceiling in that case, so the effective limit is min(requested_budget, team.max_budget) rather than unbounded. --- litellm/proxy/_types.py | 1 + litellm/proxy/auth/auth_checks.py | 1 + .../key_management_endpoints.py | 25 ++++--- .../proxy/auth/test_auth_checks.py | 3 + .../test_key_management_endpoints.py | 66 +++++++++++++++++++ 5 files changed, 87 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 5bba842c7eb..35f9a300541 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2522,6 +2522,7 @@ class UserAPIKeyAuth( user_spend: Optional[float] = None user_max_budget: Optional[float] = None request_route: Optional[str] = None + is_session_token: bool = False budget_reservation: Optional[Dict[str, Any]] = Field(default=None, exclude=True) user: Optional[Any] = None # Expanded user object when expand=user is used created_by_user: Optional[Any] = ( diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 6c8991548a9..267d87965e4 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -2471,6 +2471,7 @@ class ExperimentalUIJWTToken: models=user_info.models, max_parallel_requests=None, user_role=LitellmUserRoles(user_info.user_role), + is_session_token=True, ) return encrypt_value_helper(valid_token.model_dump_json(exclude_none=True)) diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 2d49297c8e9..719ed503ec8 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -740,29 +740,36 @@ async def _common_key_generation_helper( _enforce_upperbound_key_params(data, fill_defaults=True) # Delegated-authority ceiling (GHSA-q775-qw9r-2r4g): a non-admin caller - # with an explicit budget cannot grant a key a higher budget than their own. - # Callers with max_budget=None (unlimited) can delegate any budget. - # A UI/CLI session token's max_budget is a per-session chat spend cap - # (max_ui_session_budget), not a delegation authority, so it is exempt only - # when creating a team key - that key's spend is bounded by the team budget - # at request time. Personal keys keep the ceiling; nothing else bounds them. + # cannot grant a key a higher budget than their own authority. is_ui_session_team_key = ( user_api_key_dict.team_id == UI_SESSION_TOKEN_TEAM_ID and _requested_team_id is not None ) + # Session tokens (lite login) carry max_budget=None to avoid a per-session + # LLM spend cap, but that None must not be read as "unlimited delegation + # authority". Use the team budget as the effective ceiling instead. + delegation_ceiling = ( + user_api_key_dict.max_budget + if user_api_key_dict.max_budget is not None + else ( + team_table.max_budget + if user_api_key_dict.is_session_token and team_table is not None + else None + ) + ) if ( user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value and not is_ui_session_team_key and _requested_max_budget is not None - and user_api_key_dict.max_budget is not None - and _requested_max_budget > user_api_key_dict.max_budget + and delegation_ceiling is not None + and _requested_max_budget > delegation_ceiling ): raise HTTPException( status_code=400, detail={ "error": ( f"max_budget ({_requested_max_budget}) cannot exceed the caller's " - f"own max_budget ({user_api_key_dict.max_budget})." + f"own max_budget ({delegation_ceiling})." ) }, ) diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index 092fbd87b1f..6d343af5b15 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -462,6 +462,9 @@ def test_get_cli_jwt_auth_token_default_expiration(valid_sso_user_defined_values # CLI session tokens carry no per-key budget; spend is enforced via the # shared team/user counters. The $0.25 UI session cap must not leak in. assert token_data.get("max_budget") is None + # is_session_token=True causes key_management_endpoints to use the team + # budget as the delegation ceiling instead of treating None as unlimited. + assert token_data.get("is_session_token") is True # Verify expiration time is set to 24 hours (default) assert "expires" in token_data diff --git a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py index b8ec8a8a388..a3725e55075 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py @@ -12570,3 +12570,69 @@ async def test_list_keys_non_admin_cannot_opt_into_substring(): ) assert kwargs["use_substring_matching"] is False assert kwargs["user_id"] == "alice" + + +@pytest.mark.asyncio +async def test_cli_session_token_delegation_ceiling_blocked_by_team_budget(): + team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + team_id="team-1", + is_session_token=True, + ) + with pytest.raises(HTTPException) as exc_info: + await _common_key_generation_helper( + data=GenerateKeyRequest(max_budget=1000.0), + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=team, + ) + assert exc_info.value.status_code == 400 + assert "max_budget" in str(exc_info.value.detail) + + +@pytest.mark.asyncio +async def test_cli_session_token_delegation_allowed_within_team_budget(): + team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + team_id="team-1", + is_session_token=True, + ) + with patch( + "litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn", + new_callable=AsyncMock, + return_value={"key": "sk-test", "expires": None, "user_id": "user-1"}, + ): + result = await _common_key_generation_helper( + data=GenerateKeyRequest(max_budget=25.0), + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=team, + ) + assert result is not None + + +@pytest.mark.asyncio +async def test_regular_unlimited_user_delegation_ceiling_not_applied(): + team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + team_id="team-1", + is_session_token=False, + ) + with patch( + "litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn", + new_callable=AsyncMock, + return_value={"key": "sk-test", "expires": None, "user_id": "user-1"}, + ): + result = await _common_key_generation_helper( + data=GenerateKeyRequest(max_budget=1000.0), + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=team, + ) + assert result is not None