mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(sso): stop stamping the UI session budget on CLI login tokens (#33312)
A `lite login` token 429'd with "Budget has been exceeded! Max budget: 0.25" even when no budget was configured anywhere. cli_poll_key stamped the minted CLI session token with litellm.max_ui_session_budget ($0.25) as a fallback whenever the user and team had no budget of their own. That cap was designed for the Admin UI "Test Key" chat pane; the CLI reused the same session-token machinery, so it inherited a playground-sized budget baked into the encrypted token at login (unchangeable without re-login), which trips fast under real CLI/agent use. The cap is also redundant: the token already carries user_id and team_id, so the real user/team budgets are enforced independently at request time. Pass max_budget=None so the CLI token is governed only by those real budgets, and drop the now-dead user/team budget lookups. The UI login token's guard (get_experimental_ui_login_jwt_auth_token) is untouched.
This commit is contained in:
parent
68f0fb0346
commit
9b6289e497
2 changed files with 9 additions and 62 deletions
|
|
@ -2092,12 +2092,8 @@ async def cli_poll_key(
|
|||
key_id: The CLI login session ID
|
||||
team_id: Optional team ID to assign to the JWT. If provided, must be one of user's teams.
|
||||
"""
|
||||
from litellm.proxy.auth.auth_checks import (
|
||||
ExperimentalUIJWTToken,
|
||||
get_team_object,
|
||||
get_user_object,
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
||||
from litellm.proxy.auth.auth_checks import ExperimentalUIJWTToken
|
||||
from litellm.proxy.proxy_server import user_api_key_cache
|
||||
|
||||
try:
|
||||
flow = _get_cli_sso_flow_or_raise(login_id=key_id, cache=user_api_key_cache)
|
||||
|
|
@ -2167,43 +2163,11 @@ async def cli_poll_key(
|
|||
models=session_data.get("models", []),
|
||||
)
|
||||
|
||||
try:
|
||||
user_db_obj = await get_user_object(
|
||||
user_id=user_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
user_id_upsert=False,
|
||||
)
|
||||
except ValueError as e:
|
||||
verbose_proxy_logger.debug(f"CLI poll: user lookup failed, proceeding without user budget: {e}")
|
||||
user_db_obj = None
|
||||
user_budget = user_db_obj.max_budget if user_db_obj is not None else None
|
||||
|
||||
team_budget: Optional[float] = None
|
||||
team_budget_resolved = False
|
||||
if team_id is not None:
|
||||
try:
|
||||
team_obj = await get_team_object(
|
||||
team_id=team_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
)
|
||||
team_budget = team_obj.max_budget
|
||||
team_budget_resolved = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
session_max_budget = (
|
||||
litellm.max_ui_session_budget
|
||||
if user_budget is None and (team_id is None or (team_budget_resolved and team_budget is None))
|
||||
else None
|
||||
)
|
||||
|
||||
jwt_token = ExperimentalUIJWTToken.get_cli_jwt_auth_token(
|
||||
user_info=user_info,
|
||||
team_id=team_id,
|
||||
team_alias=team_alias,
|
||||
max_budget=session_max_budget,
|
||||
max_budget=None,
|
||||
)
|
||||
|
||||
# Delete cache entry (single-use)
|
||||
|
|
|
|||
|
|
@ -3122,9 +3122,11 @@ class TestCLIKeyRegenerationFlow:
|
|||
assert mock_get_jwt.call_args.kwargs["max_budget"] is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cli_poll_key_caps_session_when_user_and_team_have_no_budget(self):
|
||||
"""With no user and no team budget, the session falls back to max_ui_session_budget."""
|
||||
from litellm.proxy._types import LiteLLM_TeamTableCachedObj, LiteLLM_UserTable
|
||||
async def test_cli_poll_key_does_not_cap_session_even_without_user_or_team_budget(self):
|
||||
"""Regression: a CLI session token must not inherit the UI chat-pane budget
|
||||
(max_ui_session_budget). Even when the user and team have no budget of their
|
||||
own, the minted token carries max_budget=None and is governed only by the
|
||||
real user/team budgets at request time."""
|
||||
from litellm.proxy.management_endpoints.ui_sso import (
|
||||
_hash_cli_sso_secret,
|
||||
cli_poll_key,
|
||||
|
|
@ -3138,14 +3140,6 @@ class TestCLIKeyRegenerationFlow:
|
|||
"models": ["gpt-4"],
|
||||
"user_email": "unbudgeted@example.com",
|
||||
}
|
||||
mock_user_info = LiteLLM_UserTable(
|
||||
user_id="unbudgeted-user",
|
||||
user_role="internal_user",
|
||||
teams=["team-x"],
|
||||
models=["gpt-4"],
|
||||
max_budget=None,
|
||||
)
|
||||
mock_team = LiteLLM_TeamTableCachedObj(team_id="team-x", max_budget=None)
|
||||
mock_cache = MagicMock()
|
||||
mock_cache.get_cache.return_value = {
|
||||
"poll_secret_hash": _hash_cli_sso_secret("poll-secret"),
|
||||
|
|
@ -3157,19 +3151,10 @@ class TestCLIKeyRegenerationFlow:
|
|||
|
||||
with (
|
||||
patch("litellm.proxy.proxy_server.user_api_key_cache", mock_cache),
|
||||
patch("litellm.proxy.proxy_server.prisma_client"),
|
||||
patch(
|
||||
"litellm.proxy.auth.auth_checks.ExperimentalUIJWTToken.get_cli_jwt_auth_token",
|
||||
return_value=mock_jwt_token,
|
||||
) as mock_get_jwt,
|
||||
patch(
|
||||
"litellm.proxy.auth.auth_checks.get_user_object",
|
||||
new=AsyncMock(return_value=mock_user_info),
|
||||
),
|
||||
patch(
|
||||
"litellm.proxy.auth.auth_checks.get_team_object",
|
||||
new=AsyncMock(return_value=mock_team),
|
||||
),
|
||||
):
|
||||
result = await cli_poll_key(
|
||||
key_id="cli-session-unbudgeted",
|
||||
|
|
@ -3179,9 +3164,7 @@ class TestCLIKeyRegenerationFlow:
|
|||
|
||||
assert result["status"] == "ready"
|
||||
mock_get_jwt.assert_called_once()
|
||||
assert (
|
||||
mock_get_jwt.call_args.kwargs["max_budget"] == litellm.max_ui_session_budget
|
||||
)
|
||||
assert mock_get_jwt.call_args.kwargs["max_budget"] is None
|
||||
|
||||
|
||||
class TestGetAppRolesFromIdToken:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue