Merge pull request #37044 from Thijmen/key-budget-window-usage

feat(key management): show budget window usage on /key/info
This commit is contained in:
ryan-crabbe-berri 2026-08-31 21:18:46 -07:00 committed by GitHub
commit fa720be1f4
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 385 additions and 1 deletions

View file

@ -20,6 +20,7 @@ import secrets
import traceback
from collections.abc import Awaitable, Callable, Iterator, Mapping, Sequence
from datetime import datetime, timedelta, timezone
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Literal, Optional, Protocol, TypeVar, cast
import fastapi
@ -111,6 +112,7 @@ from litellm.proxy.management_helpers.team_member_permission_checks import (
TeamMemberPermissionChecks,
)
from litellm.proxy.management_helpers.utils import management_endpoint_wrapper
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.spend_tracking_utils import _is_master_key
from litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints import (
get_ui_settings_cached,
@ -3578,6 +3580,63 @@ async def _build_model_max_budget_usage(
)
def _window_max_budget(window: Mapping[str, object]) -> float | None:
"""A window's max_budget as a float; None when absent or unparseable."""
value: Final = window.get("max_budget")
if not isinstance(value, (int, float, str)):
return None
try:
return float(value)
except ValueError:
return None
async def _budget_window_usage(
window: Mapping[str, object], api_key_hash: str
) -> tuple[str, Mapping[str, object]] | None:
"""
(budget_duration, usage entry) for one budget window; None when the window
has no budget_duration to key it by.
Reads the same cross-pod counter (spend:key:{hashed_token}:window:{budget_duration})
that _virtual_key_multi_budget_check enforces against, passing the same
window_duration + window_start so a stale-low counter is re-checked against
the LiteLLM_BudgetWindowSpend row instead of a spend-log aggregate.
"""
from litellm.proxy.proxy_server import get_current_spend
duration: Final = window.get("budget_duration")
if not isinstance(duration, str) or not duration:
return None
spend: Final = await get_current_spend(
counter_key=f"spend:key:{api_key_hash}:window:{duration}",
fallback_spend=0.0,
max_budget=_window_max_budget(window),
window_entity_type="Key",
window_entity_id=api_key_hash,
window_duration=duration,
window_start=get_budget_window_start(window),
)
return duration, MappingProxyType({"current_spend": round(spend, 4)})
async def _build_budget_limits_usage(
budget_limits: Sequence[object] | str | None, api_key_hash: str
) -> Mapping[str, Mapping[str, object]] | None:
"""
Current-window spend per budget window, keyed by budget_duration, reported
next to the stored budget_limits (which is returned untouched). None when
the key has no windows, so the field only appears on keys that have them.
"""
windows: Final = _budget_limit_windows(budget_limits)
if not windows:
return None
usages: Final = await asyncio.gather(
*(_budget_window_usage(window=window, api_key_hash=api_key_hash) for window in windows)
)
return MappingProxyType({duration: usage for duration, usage in (u for u in usages if u is not None)})
@router.post(
"/v2/key/info",
tags=["key management"],
@ -3620,7 +3679,6 @@ async def info_key_fn_v2(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail={"message": "Malformed request. No keys passed in."},
)
# Resolve key_aliases to tokens so we never pass token=None (unbounded query)
tokens_to_query: Final = list(data.keys) if data.keys else []
if data.key_aliases:
@ -3662,6 +3720,13 @@ async def info_key_fn_v2(
model_max_budget=model_max_budget,
user_api_key_cache=model_max_budget_limiter.dual_cache,
)
if k_token_hash:
budget_limits_usage = await _build_budget_limits_usage(
budget_limits=k_dict.get("budget_limits"),
api_key_hash=k_token_hash,
)
if budget_limits_usage is not None:
k_dict["budget_limits_usage"] = budget_limits_usage
filtered_key_info.append(k_dict)
return {"key": data.keys, "info": filtered_key_info}
@ -3698,6 +3763,10 @@ async def info_key_fn(
- model_max_budget: dict - Per-model budgets, e.g. {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}
- model_max_budget_usage: dict | None - Current-window spend per model, present only when
the key has per-model budgets
- budget_limits: list | None - Concurrent budget windows, exactly as stored
- budget_limits_usage: dict | None - Current-window spend per budget window, e.g.
{"1h": {"current_spend": 0.0009}}, present only when the key has budget windows
(read from the same cross-pod spend counter the budget enforcement uses)
- models: list - Model_name's the key is allowed to call
- tpm_limit / rpm_limit: int | None - Tokens and requests per minute limits
- metadata: dict - Metadata for the key, e.g. {"team": "core-infra"}
@ -3777,6 +3846,12 @@ async def info_key_fn(
model_max_budget=model_max_budget,
user_api_key_cache=model_max_budget_limiter.dual_cache,
)
budget_limits_usage: Final = await _build_budget_limits_usage(
budget_limits=key_info.get("budget_limits"),
api_key_hash=key_token_hash,
)
if budget_limits_usage is not None:
key_info["budget_limits_usage"] = budget_limits_usage
# Attach object_permission if object_permission_id is set
key_info = await attach_object_permission_to_dict(key_info, prisma_client)

View file

@ -14142,6 +14142,311 @@ async def test_info_key_fn_v2_budget_table_fallback(monkeypatch):
mock_prisma_client.db.query_raw.assert_not_awaited()
@pytest.mark.asyncio
async def test_info_key_fn_reports_budget_limits_usage(monkeypatch):
"""
/key/info reports current-window spend per budget window under budget_limits_usage,
keyed by budget_duration and read from the same counter enforcement uses, while
budget_limits itself comes back exactly as stored.
"""
from unittest.mock import AsyncMock, MagicMock
from litellm.proxy._types import LiteLLM_VerificationToken
from litellm.proxy.management_endpoints.key_management_endpoints import info_key_fn
test_key_token = "hashed_token_window_test"
budget_limits = [
{
"reset_at": "2026-08-15T18:00:00+00:00",
"max_budget": 2.0,
"budget_duration": "1h",
}
]
mock_prisma_client = AsyncMock()
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_user_api_key_cache = AsyncMock()
monkeypatch.setattr(
"litellm.proxy.proxy_server.user_api_key_cache", mock_user_api_key_cache
)
mock_get_current_spend = AsyncMock(return_value=0.73)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
mock_key_info = MagicMock(spec=LiteLLM_VerificationToken)
mock_key_info.token = test_key_token
mock_key_info.object_permission_id = None
mock_key_info.user_id = "user-w"
mock_key_info.team_id = None
mock_key_info.litellm_budget_table = None
mock_key_info.model_dump.return_value = {
"token": test_key_token,
"budget_limits": [dict(w) for w in budget_limits],
"user_id": "user-w",
"team_id": None,
"object_permission_id": None,
"litellm_budget_table": None,
}
mock_key_info.dict.return_value = mock_key_info.model_dump.return_value
mock_prisma_client.db.litellm_verificationtoken.find_unique = AsyncMock(
return_value=mock_key_info
)
user_api_key_dict = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN,
api_key="sk-test-window-key",
)
result = await info_key_fn(
key="sk-test-window-key",
user_api_key_dict=user_api_key_dict,
)
assert result["info"]["budget_limits"] == budget_limits
assert result["info"]["budget_limits_usage"] == {"1h": {"current_spend": 0.73}}
mock_get_current_spend.assert_awaited_once()
call_kwargs = mock_get_current_spend.await_args.kwargs
assert call_kwargs["counter_key"] == f"spend:key:{test_key_token}:window:1h"
assert call_kwargs["max_budget"] == 2.0
assert call_kwargs["window_entity_type"] == "Key"
assert call_kwargs["window_entity_id"] == test_key_token
assert call_kwargs["window_duration"] == "1h"
assert call_kwargs["window_start"] is not None
@pytest.mark.asyncio
async def test_info_key_fn_no_budget_limits_skips_spend_lookup(monkeypatch):
"""Keys without budget windows get no budget_limits_usage field and trigger no spend lookup."""
from unittest.mock import AsyncMock, MagicMock
from litellm.proxy._types import LiteLLM_VerificationToken
from litellm.proxy.management_endpoints.key_management_endpoints import info_key_fn
test_key_token = "hashed_token_no_windows"
mock_prisma_client = AsyncMock()
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_user_api_key_cache = AsyncMock()
monkeypatch.setattr(
"litellm.proxy.proxy_server.user_api_key_cache", mock_user_api_key_cache
)
mock_get_current_spend = AsyncMock(return_value=0.0)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
mock_key_info = MagicMock(spec=LiteLLM_VerificationToken)
mock_key_info.token = test_key_token
mock_key_info.object_permission_id = None
mock_key_info.user_id = "user-nw"
mock_key_info.team_id = None
mock_key_info.litellm_budget_table = None
mock_key_info.model_dump.return_value = {
"token": test_key_token,
"budget_limits": None,
"user_id": "user-nw",
"team_id": None,
"object_permission_id": None,
"litellm_budget_table": None,
}
mock_key_info.dict.return_value = mock_key_info.model_dump.return_value
mock_prisma_client.db.litellm_verificationtoken.find_unique = AsyncMock(
return_value=mock_key_info
)
user_api_key_dict = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN,
api_key="sk-test-no-window-key",
)
result = await info_key_fn(
key="sk-test-no-window-key",
user_api_key_dict=user_api_key_dict,
)
assert result["info"]["budget_limits"] is None
assert "budget_limits_usage" not in result["info"]
mock_get_current_spend.assert_not_awaited()
@pytest.mark.asyncio
async def test_info_key_fn_v2_reports_budget_limits_usage(monkeypatch):
"""/v2/key/info reports budget_limits_usage per window and leaves budget_limits as stored."""
from unittest.mock import AsyncMock, MagicMock
from litellm.proxy._types import KeyRequest, LiteLLM_VerificationToken
from litellm.proxy.management_endpoints.key_management_endpoints import (
info_key_fn_v2,
)
test_key_token = "hashed_token_v2_window_test"
budget_limits = [
{
"reset_at": "2026-08-15T18:00:00+00:00",
"max_budget": 2.0,
"budget_duration": "1h",
},
{
"reset_at": "2026-08-16T00:00:00+00:00",
"max_budget": 20.0,
"budget_duration": "1d",
},
]
mock_prisma_client = AsyncMock()
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_user_api_key_cache = AsyncMock()
monkeypatch.setattr(
"litellm.proxy.proxy_server.user_api_key_cache", mock_user_api_key_cache
)
mock_get_current_spend = AsyncMock(return_value=1.25)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
mock_key = MagicMock(spec=LiteLLM_VerificationToken)
mock_key.token = test_key_token
mock_key.user_id = "user-v2-w"
mock_key.team_id = None
mock_key.model_dump.return_value = {
"token": test_key_token,
"budget_limits": [dict(w) for w in budget_limits],
"user_id": "user-v2-w",
"team_id": None,
"litellm_budget_table": None,
}
mock_key.dict.return_value = mock_key.model_dump.return_value
mock_prisma_client.get_data = AsyncMock(return_value=[mock_key])
user_api_key_dict = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN,
api_key="sk-admin-v2-w",
)
result = await info_key_fn_v2(
data=KeyRequest(keys=[test_key_token]),
user_api_key_dict=user_api_key_dict,
)
assert len(result["info"]) == 1
assert result["info"][0]["budget_limits"] == budget_limits
assert result["info"][0]["budget_limits_usage"] == {
"1h": {"current_spend": 1.25},
"1d": {"current_spend": 1.25},
}
assert mock_get_current_spend.await_count == 2
counter_keys = {
call.kwargs["counter_key"] for call in mock_get_current_spend.await_args_list
}
assert counter_keys == {
f"spend:key:{test_key_token}:window:1h",
f"spend:key:{test_key_token}:window:1d",
}
assert {
call.kwargs["window_duration"] for call in mock_get_current_spend.await_args_list
} == {"1h", "1d"}
@pytest.mark.asyncio
async def test_build_budget_limits_usage_json_string_input(monkeypatch):
"""budget_limits stored as a JSON string is parsed and reported per window."""
import json as json_module
from unittest.mock import AsyncMock
from litellm.proxy.management_endpoints.key_management_endpoints import (
_build_budget_limits_usage,
)
mock_get_current_spend = AsyncMock(return_value=0.5)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
raw = json_module.dumps(
[{"budget_duration": "1h", "max_budget": 2.0, "reset_at": None}]
)
result = await _build_budget_limits_usage(budget_limits=raw, api_key_hash="hash-1")
assert result == {"1h": {"current_spend": 0.5}}
mock_get_current_spend.assert_awaited_once()
@pytest.mark.asyncio
async def test_build_budget_limits_usage_empty_windows_returns_none(monkeypatch):
"""A key with no windows (None, [], or "[]") returns None so the field is left off; no spend lookup runs."""
from unittest.mock import AsyncMock
from litellm.proxy.management_endpoints.key_management_endpoints import (
_build_budget_limits_usage,
)
mock_get_current_spend = AsyncMock(return_value=0.0)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
for stored in (None, [], "[]"):
assert await _build_budget_limits_usage(budget_limits=stored, api_key_hash="hash-1") is None
mock_get_current_spend.assert_not_awaited()
@pytest.mark.asyncio
async def test_build_budget_limits_usage_window_without_max_budget(monkeypatch):
"""A window with only budget_duration still reports current_spend, read without a budget ceiling."""
from unittest.mock import AsyncMock
from litellm.proxy.management_endpoints.key_management_endpoints import (
_build_budget_limits_usage,
)
mock_get_current_spend = AsyncMock(return_value=0.75)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
result = await _build_budget_limits_usage(
budget_limits=[{"budget_duration": "2d"}], api_key_hash="hash-no-max"
)
assert result == {"2d": {"current_spend": 0.75}}
call_kwargs = mock_get_current_spend.await_args.kwargs
assert call_kwargs["counter_key"] == "spend:key:hash-no-max:window:2d"
assert call_kwargs["window_duration"] == "2d"
assert call_kwargs["max_budget"] is None
@pytest.mark.asyncio
async def test_build_budget_limits_usage_pydantic_windows(monkeypatch):
"""BudgetLimitEntry windows (the shape UserAPIKeyAuth carries) are dumped to dicts and reported."""
from unittest.mock import AsyncMock
from litellm.models.team import BudgetLimitEntry
from litellm.proxy.management_endpoints.key_management_endpoints import (
_build_budget_limits_usage,
)
mock_get_current_spend = AsyncMock(return_value=1.0)
monkeypatch.setattr(
"litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend
)
result = await _build_budget_limits_usage(
budget_limits=[BudgetLimitEntry(budget_duration="7d", max_budget=10.0)],
api_key_hash="hash-2",
)
assert result == {"7d": {"current_spend": 1.0}}
call_kwargs = mock_get_current_spend.await_args.kwargs
assert call_kwargs["counter_key"] == "spend:key:hash-2:window:7d"
assert call_kwargs["window_duration"] == "7d"
assert call_kwargs["max_budget"] == 10.0
@pytest.mark.asyncio
async def test_info_key_fn_reads_the_configured_budget_model_key(monkeypatch):
"""/key/info reads the one counter enforcement reads: the configured budget model.

View file

@ -7710,6 +7710,10 @@ export interface paths {
* - model_max_budget: dict - Per-model budgets, e.g. {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}
* - model_max_budget_usage: dict | None - Current-window spend per model, present only when
* the key has per-model budgets
* - budget_limits: list | None - Concurrent budget windows, exactly as stored
* - budget_limits_usage: dict | None - Current-window spend per budget window, e.g.
* {"1h": {"current_spend": 0.0009}}, present only when the key has budget windows
* (read from the same cross-pod spend counter the budget enforcement uses)
* - models: list - Model_name's the key is allowed to call
* - tpm_limit / rpm_limit: int | None - Tokens and requests per minute limits
* - metadata: dict - Metadata for the key, e.g. {"team": "core-infra"}