mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
refactor(key): trim budget_limits_usage entries to current_spend
max_budget and reset_at already live on the matching budget_limits entry, so repeating them (as budget_limit and reset_at) only invited confusion about which copy is authoritative.
This commit is contained in:
parent
46d073b26f
commit
760b864e43
3 changed files with 14 additions and 35 deletions
|
|
@ -3608,23 +3608,16 @@ async def _budget_window_usage(
|
|||
duration: Final = window.get("budget_duration")
|
||||
if not isinstance(duration, str) or not duration:
|
||||
return None
|
||||
max_budget: Final = _window_max_budget(window)
|
||||
spend: Final = await get_current_spend(
|
||||
counter_key=f"spend:key:{api_key_hash}:window:{duration}",
|
||||
fallback_spend=0.0,
|
||||
max_budget=max_budget,
|
||||
max_budget=_window_max_budget(window),
|
||||
window_entity_type="Key",
|
||||
window_entity_id=api_key_hash,
|
||||
window_duration=duration,
|
||||
window_start=get_budget_window_start(window),
|
||||
)
|
||||
return duration, MappingProxyType(
|
||||
{
|
||||
"current_spend": round(spend, 4),
|
||||
"budget_limit": max_budget,
|
||||
"reset_at": window.get("reset_at"),
|
||||
}
|
||||
)
|
||||
return duration, MappingProxyType({"current_spend": round(spend, 4)})
|
||||
|
||||
|
||||
async def _build_budget_limits_usage(
|
||||
|
|
@ -3771,9 +3764,9 @@ async def info_key_fn(
|
|||
- model_max_budget_usage: dict | None - Current-window spend per model, present only when
|
||||
the key has per-model budgets
|
||||
- budget_limits: list | None - Concurrent budget windows, exactly as stored
|
||||
- budget_limits_usage: dict | None - Current-window spend per budget window, keyed by
|
||||
budget_duration, present only when the key has budget windows (read from the same
|
||||
cross-pod spend counter the budget enforcement uses)
|
||||
- budget_limits_usage: dict | None - Current-window spend per budget window, e.g.
|
||||
{"1h": {"current_spend": 0.0009}}, present only when the key has budget windows
|
||||
(read from the same cross-pod spend counter the budget enforcement uses)
|
||||
- models: list - Model_name's the key is allowed to call
|
||||
- tpm_limit / rpm_limit: int | None - Tokens and requests per minute limits
|
||||
- metadata: dict - Metadata for the key, e.g. {"team": "core-infra"}
|
||||
|
|
|
|||
|
|
@ -14205,13 +14205,7 @@ async def test_info_key_fn_reports_budget_limits_usage(monkeypatch):
|
|||
)
|
||||
|
||||
assert result["info"]["budget_limits"] == budget_limits
|
||||
assert result["info"]["budget_limits_usage"] == {
|
||||
"1h": {
|
||||
"current_spend": 0.73,
|
||||
"budget_limit": 2.0,
|
||||
"reset_at": "2026-08-15T18:00:00+00:00",
|
||||
}
|
||||
}
|
||||
assert result["info"]["budget_limits_usage"] == {"1h": {"current_spend": 0.73}}
|
||||
|
||||
mock_get_current_spend.assert_awaited_once()
|
||||
call_kwargs = mock_get_current_spend.await_args.kwargs
|
||||
|
|
@ -14342,16 +14336,8 @@ async def test_info_key_fn_v2_reports_budget_limits_usage(monkeypatch):
|
|||
assert len(result["info"]) == 1
|
||||
assert result["info"][0]["budget_limits"] == budget_limits
|
||||
assert result["info"][0]["budget_limits_usage"] == {
|
||||
"1h": {
|
||||
"current_spend": 1.25,
|
||||
"budget_limit": 2.0,
|
||||
"reset_at": "2026-08-15T18:00:00+00:00",
|
||||
},
|
||||
"1d": {
|
||||
"current_spend": 1.25,
|
||||
"budget_limit": 20.0,
|
||||
"reset_at": "2026-08-16T00:00:00+00:00",
|
||||
},
|
||||
"1h": {"current_spend": 1.25},
|
||||
"1d": {"current_spend": 1.25},
|
||||
}
|
||||
assert mock_get_current_spend.await_count == 2
|
||||
counter_keys = {
|
||||
|
|
@ -14386,7 +14372,7 @@ async def test_build_budget_limits_usage_json_string_input(monkeypatch):
|
|||
)
|
||||
result = await _build_budget_limits_usage(budget_limits=raw, api_key_hash="hash-1")
|
||||
|
||||
assert result == {"1h": {"current_spend": 0.5, "budget_limit": 2.0, "reset_at": None}}
|
||||
assert result == {"1h": {"current_spend": 0.5}}
|
||||
mock_get_current_spend.assert_awaited_once()
|
||||
|
||||
|
||||
|
|
@ -14427,7 +14413,7 @@ async def test_build_budget_limits_usage_window_without_max_budget(monkeypatch):
|
|||
budget_limits=[{"budget_duration": "2d"}], api_key_hash="hash-no-max"
|
||||
)
|
||||
|
||||
assert result == {"2d": {"current_spend": 0.75, "budget_limit": None, "reset_at": None}}
|
||||
assert result == {"2d": {"current_spend": 0.75}}
|
||||
call_kwargs = mock_get_current_spend.await_args.kwargs
|
||||
assert call_kwargs["counter_key"] == "spend:key:hash-no-max:window:2d"
|
||||
assert call_kwargs["window_duration"] == "2d"
|
||||
|
|
@ -14454,7 +14440,7 @@ async def test_build_budget_limits_usage_pydantic_windows(monkeypatch):
|
|||
api_key_hash="hash-2",
|
||||
)
|
||||
|
||||
assert result == {"7d": {"current_spend": 1.0, "budget_limit": 10.0, "reset_at": None}}
|
||||
assert result == {"7d": {"current_spend": 1.0}}
|
||||
call_kwargs = mock_get_current_spend.await_args.kwargs
|
||||
assert call_kwargs["counter_key"] == "spend:key:hash-2:window:7d"
|
||||
assert call_kwargs["window_duration"] == "7d"
|
||||
|
|
|
|||
6
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
6
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -7711,9 +7711,9 @@ export interface paths {
|
|||
* - model_max_budget_usage: dict | None - Current-window spend per model, present only when
|
||||
* the key has per-model budgets
|
||||
* - budget_limits: list | None - Concurrent budget windows, exactly as stored
|
||||
* - budget_limits_usage: dict | None - Current-window spend per budget window, keyed by
|
||||
* budget_duration, present only when the key has budget windows (read from the same
|
||||
* cross-pod spend counter the budget enforcement uses)
|
||||
* - budget_limits_usage: dict | None - Current-window spend per budget window, e.g.
|
||||
* {"1h": {"current_spend": 0.0009}}, present only when the key has budget windows
|
||||
* (read from the same cross-pod spend counter the budget enforcement uses)
|
||||
* - models: list - Model_name's the key is allowed to call
|
||||
* - tpm_limit / rpm_limit: int | None - Tokens and requests per minute limits
|
||||
* - metadata: dict - Metadata for the key, e.g. {"team": "core-infra"}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue