fix(key-management): cap CLI session token delegation budget to team ceiling

A CLI session token intentionally carries max_budget=None to avoid a per-session LLM spend cap. The key-generation delegation check (GHSA-q775-qw9r-2r4g) previously skipped non-admin callers with max_budget=None, treating them as having unlimited delegation authority. This allowed any internal user with a lite login session to mint virtual keys with arbitrary budgets.

Adds is_session_token=True to UserAPIKeyAuth for CLI session tokens and uses the caller's team budget as the delegation ceiling in that case, so the effective limit is min(requested_budget, team.max_budget) rather than unbounded.
This commit is contained in:
Sameer Kankute 2026-06-23 17:53:26 +05:30
parent ef8482dcd4
commit 91e60b80b0
No known key found for this signature in database
5 changed files with 87 additions and 9 deletions

View file

@ -2522,6 +2522,7 @@ class UserAPIKeyAuth(
user_spend: Optional[float] = None
user_max_budget: Optional[float] = None
request_route: Optional[str] = None
is_session_token: bool = False
budget_reservation: Optional[Dict[str, Any]] = Field(default=None, exclude=True)
user: Optional[Any] = None # Expanded user object when expand=user is used
created_by_user: Optional[Any] = (

View file

@ -2471,6 +2471,7 @@ class ExperimentalUIJWTToken:
models=user_info.models,
max_parallel_requests=None,
user_role=LitellmUserRoles(user_info.user_role),
is_session_token=True,
)
return encrypt_value_helper(valid_token.model_dump_json(exclude_none=True))

View file

@ -740,29 +740,36 @@ async def _common_key_generation_helper(
_enforce_upperbound_key_params(data, fill_defaults=True)
# Delegated-authority ceiling (GHSA-q775-qw9r-2r4g): a non-admin caller
# with an explicit budget cannot grant a key a higher budget than their own.
# Callers with max_budget=None (unlimited) can delegate any budget.
# A UI/CLI session token's max_budget is a per-session chat spend cap
# (max_ui_session_budget), not a delegation authority, so it is exempt only
# when creating a team key - that key's spend is bounded by the team budget
# at request time. Personal keys keep the ceiling; nothing else bounds them.
# cannot grant a key a higher budget than their own authority.
is_ui_session_team_key = (
user_api_key_dict.team_id == UI_SESSION_TOKEN_TEAM_ID
and _requested_team_id is not None
)
# Session tokens (lite login) carry max_budget=None to avoid a per-session
# LLM spend cap, but that None must not be read as "unlimited delegation
# authority". Use the team budget as the effective ceiling instead.
delegation_ceiling = (
user_api_key_dict.max_budget
if user_api_key_dict.max_budget is not None
else (
team_table.max_budget
if user_api_key_dict.is_session_token and team_table is not None
else None
)
)
if (
user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN.value
and not is_ui_session_team_key
and _requested_max_budget is not None
and user_api_key_dict.max_budget is not None
and _requested_max_budget > user_api_key_dict.max_budget
and delegation_ceiling is not None
and _requested_max_budget > delegation_ceiling
):
raise HTTPException(
status_code=400,
detail={
"error": (
f"max_budget ({_requested_max_budget}) cannot exceed the caller's "
f"own max_budget ({user_api_key_dict.max_budget})."
f"own max_budget ({delegation_ceiling})."
)
},
)

View file

@ -462,6 +462,9 @@ def test_get_cli_jwt_auth_token_default_expiration(valid_sso_user_defined_values
# CLI session tokens carry no per-key budget; spend is enforced via the
# shared team/user counters. The $0.25 UI session cap must not leak in.
assert token_data.get("max_budget") is None
# is_session_token=True causes key_management_endpoints to use the team
# budget as the delegation ceiling instead of treating None as unlimited.
assert token_data.get("is_session_token") is True
# Verify expiration time is set to 24 hours (default)
assert "expires" in token_data

View file

@ -12570,3 +12570,69 @@ async def test_list_keys_non_admin_cannot_opt_into_substring():
)
assert kwargs["use_substring_matching"] is False
assert kwargs["user_id"] == "alice"
@pytest.mark.asyncio
async def test_cli_session_token_delegation_ceiling_blocked_by_team_budget():
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
caller = UserAPIKeyAuth(
user_role=LitellmUserRoles.INTERNAL_USER,
user_id="user-1",
team_id="team-1",
is_session_token=True,
)
with pytest.raises(HTTPException) as exc_info:
await _common_key_generation_helper(
data=GenerateKeyRequest(max_budget=1000.0),
user_api_key_dict=caller,
litellm_changed_by=None,
team_table=team,
)
assert exc_info.value.status_code == 400
assert "max_budget" in str(exc_info.value.detail)
@pytest.mark.asyncio
async def test_cli_session_token_delegation_allowed_within_team_budget():
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
caller = UserAPIKeyAuth(
user_role=LitellmUserRoles.INTERNAL_USER,
user_id="user-1",
team_id="team-1",
is_session_token=True,
)
with patch(
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn",
new_callable=AsyncMock,
return_value={"key": "sk-test", "expires": None, "user_id": "user-1"},
):
result = await _common_key_generation_helper(
data=GenerateKeyRequest(max_budget=25.0),
user_api_key_dict=caller,
litellm_changed_by=None,
team_table=team,
)
assert result is not None
@pytest.mark.asyncio
async def test_regular_unlimited_user_delegation_ceiling_not_applied():
team = LiteLLM_TeamTableCachedObj(team_id="team-1", max_budget=50.0)
caller = UserAPIKeyAuth(
user_role=LitellmUserRoles.INTERNAL_USER,
user_id="user-1",
team_id="team-1",
is_session_token=False,
)
with patch(
"litellm.proxy.management_endpoints.key_management_endpoints.generate_key_helper_fn",
new_callable=AsyncMock,
return_value={"key": "sk-test", "expires": None, "user_id": "user-1"},
):
result = await _common_key_generation_helper(
data=GenerateKeyRequest(max_budget=1000.0),
user_api_key_dict=caller,
litellm_changed_by=None,
team_table=team,
)
assert result is not None