mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(key-management): cap CLI session token delegation budget to team ceiling
A CLI session token intentionally carries max_budget=None to avoid a per-session LLM spend cap. The key-generation delegation check (GHSA-q775-qw9r-2r4g) previously skipped non-admin callers with max_budget=None, treating them as having unlimited delegation authority. This allowed any internal user with a lite login session to mint virtual keys with arbitrary budgets. Adds is_session_token=True to UserAPIKeyAuth for CLI session tokens and uses the caller's team budget as the delegation ceiling in that case, so the effective limit is min(requested_budget, team.max_budget) rather than unbounded.
This commit is contained in:
parent
ef8482dcd4
commit
91e60b80b0
5 changed files with 87 additions and 9 deletions
|
|
@ -2522,6 +2522,7 @@ class UserAPIKeyAuth(
|
|||
user_spend: Optional[float] = None
|
||||
user_max_budget: Optional[float] = None
|
||||
request_route: Optional[str] = None
|
||||
is_session_token: bool = False
|
||||
budget_reservation: Optional[Dict[str, Any]] = Field(default=None, exclude=True)
|
||||
user: Optional[Any] = None # Expanded user object when expand=user is used
|
||||
created_by_user: Optional[Any] = (
|
||||
|
|
|
|||
|
|
@ -2471,6 +2471,7 @@ class ExperimentalUIJWTToken:
|
|||
models=user_info.models,
|
||||
max_parallel_requests=None,
|
||||
user_role=LitellmUserRoles(user_info.user_role),
|
||||
is_session_token=True,
|
||||
)
|
||||
|
||||
return encrypt_value_helper(valid_token.model_dump_json(exclude_none=True))
|
||||
|
|
|
|||
|
|
@ -740,29 +740,36 @@ async def _common_key_generation_helper(
|
|||
_enforce_upperbound_key_params(data, fill_defaults=True)
|
||||
|
||||
# Delegated-authority ceiling (GHSA-q775-qw9r-2r4g): a non-admin caller
|
||||
# with an explicit budget cannot grant a key a higher budget than their own.
|
||||
# Callers with max_budget=None (unlimited) can delegate any budget.
|
||||
# A UI/CLI session token's max_budget is a per-session chat spend cap
|
||||
# (max_ui_session_budget), not a delegation authority, so it is exempt only
|
||||
# when creating a team key - that key's spend is bounded by the team budget
|
||||
# at request time. Personal keys keep the ceiling; nothing else bounds them.
|
||||
# cannot grant a key a higher budget than their own authority.
|
||||
is_ui_session_team_key = (
|
||||
user_api_key_dict.team_id == UI_SESSION_TOKEN_TEAM_ID
|
||||
and _requested_team_id is not None
|
||||
)
|
||||
# Session tokens (lite login) carry max_budget=None to avoid a per-session
|
||||
# LLM spend cap, but that None must not be read as "unlimited delegation
|
||||
# authority". Use the team budget as the effective ceiling instead.
|
||||
delegation_ceiling = (
|
||||
user_api_key_dict.max_budget
|
||||
if user_api_key_dict.max_budget is not None
|
||||
else (
|
||||
team_table.max_budget
|
||||
if user_api_key_dict.is_session_token and team_table is not None
|
||||
else None
|
||||
)
|
||||
)
|
||||
if (
|
||||
user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value
|
||||
and not is_ui_session_team_key
|
||||
and _requested_max_budget is not None
|
||||
and user_api_key_dict.max_budget is not None
|
||||
and _requested_max_budget > user_api_key_dict.max_budget
|
||||
and delegation_ceiling is not None
|
||||
and _requested_max_budget > delegation_ceiling
|
||||
):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={
|
||||
"error": (
|
||||
f"max_budget ({_requested_max_budget}) cannot exceed the caller's "
|
||||
f"own max_budget ({user_api_key_dict.max_budget})."
|
||||
f"own max_budget ({delegation_ceiling})."
|
||||
)
|
||||
},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -462,6 +462,9 @@ def test_get_cli_jwt_auth_token_default_expiration(valid_sso_user_defined_values
|
|||
# CLI session tokens carry no per-key budget; spend is enforced via the
|
||||
# shared team/user counters. The $0.25 UI session cap must not leak in.
|
||||
assert token_data.get("max_budget") is None
|
||||
# is_session_token=True causes key_management_endpoints to use the team
|
||||
# budget as the delegation ceiling instead of treating None as unlimited.
|
||||
assert token_data.get("is_session_token") is True
|
||||
|
||||
# Verify expiration time is set to 24 hours (default)
|
||||
assert "expires" in token_data
|
||||
|
|
|
|||
|
|
@ -12570,3 +12570,69 @@ async def test_list_keys_non_admin_cannot_opt_into_substring():
|
|||
)
|
||||
assert kwargs["use_substring_matching"] is False
|
||||
assert kwargs["user_id"] == "alice"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cli_session_token_delegation_ceiling_blocked_by_team_budget():
|
||||
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
|
||||
caller = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.INTERNAL_USER,
|
||||
user_id="user-1",
|
||||
team_id="team-1",
|
||||
is_session_token=True,
|
||||
)
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await _common_key_generation_helper(
|
||||
data=GenerateKeyRequest(max_budget=1000.0),
|
||||
user_api_key_dict=caller,
|
||||
litellm_changed_by=None,
|
||||
team_table=team,
|
||||
)
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "max_budget" in str(exc_info.value.detail)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cli_session_token_delegation_allowed_within_team_budget():
|
||||
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
|
||||
caller = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.INTERNAL_USER,
|
||||
user_id="user-1",
|
||||
team_id="team-1",
|
||||
is_session_token=True,
|
||||
)
|
||||
with patch(
|
||||
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn",
|
||||
new_callable=AsyncMock,
|
||||
return_value={"key": "sk-test", "expires": None, "user_id": "user-1"},
|
||||
):
|
||||
result = await _common_key_generation_helper(
|
||||
data=GenerateKeyRequest(max_budget=25.0),
|
||||
user_api_key_dict=caller,
|
||||
litellm_changed_by=None,
|
||||
team_table=team,
|
||||
)
|
||||
assert result is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_regular_unlimited_user_delegation_ceiling_not_applied():
|
||||
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
|
||||
caller = UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.INTERNAL_USER,
|
||||
user_id="user-1",
|
||||
team_id="team-1",
|
||||
is_session_token=False,
|
||||
)
|
||||
with patch(
|
||||
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn",
|
||||
new_callable=AsyncMock,
|
||||
return_value={"key": "sk-test", "expires": None, "user_id": "user-1"},
|
||||
):
|
||||
result = await _common_key_generation_helper(
|
||||
data=GenerateKeyRequest(max_budget=1000.0),
|
||||
user_api_key_dict=caller,
|
||||
litellm_changed_by=None,
|
||||
team_table=team,
|
||||
)
|
||||
assert result is not None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue