From 2d962c4239748dce554a3b072cab473d8bef2166 Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Wed, 19 Aug 2026 19:14:46 -0700 Subject: [PATCH] fix(key budgets): report the zero a cold per-model counter will be enforced as A per-model row whose counter has not been created yet sent a null spend, so the table showed "$0.00 of $40.00" beside a Remaining of "-": two cells contradicting each other on one row, and a client left to guess whether the null meant zero. The two non-live states are not the same kind of absence. A counter that does not exist yet is a number we know, because the check reading it treats the absence as untouched headroom and compares the cap against nothing, so zero is what will actually be enforced. Send it, with the full headroom as remaining, and let spend_state say the zero is not a reading. A failed read is a number we do not have, so it stays null in both fields; sending zero there would draw an exhausted budget as untouched. --- .../key_budget_resolver.py | 10 +++++++-- .../key_management_endpoints.py | 7 ++++--- .../test_key_management_endpoints.py | 21 +++++++++++++------ 3 files changed, 27 insertions(+), 11 deletions(-) diff --git a/litellm/proxy/management_endpoints/key_budget_resolver.py b/litellm/proxy/management_endpoints/key_budget_resolver.py index d94742c2dc1..ea33c9a1676 100644 --- a/litellm/proxy/management_endpoints/key_budget_resolver.py +++ b/litellm/proxy/management_endpoints/key_budget_resolver.py @@ -537,9 +537,15 @@ def _collapse_shared_counters(pairs: tuple[_ReadPlan, ...]) -> tuple[_ReadPlan, def _model_reading(hit: ModelSpendHit | None) -> _SpendReading: - """Per-model budgets are cache-only, so a missing counter means untouched rather than unreadable.""" + """ + A missing per-model counter is reported as no spend, because that is what will be enforced. + + The check reading it treats the absence as untouched headroom rather than as an error, so zero is + the number the cap will actually be compared against. ``spend_state`` still says the counter does + not exist, which is the part a reader needs to tell this apart from a counter that says zero. + """ if hit is None: - return _SpendReading(value=None, state="no_counter") + return _SpendReading(value=0.0, state="no_counter") return _SpendReading(value=hit.spend, state="live", counter_model=hit.model) diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index f95e35a3df2..f370169246f 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -3828,9 +3828,10 @@ async def key_budgets_fn( - max_budget: float | None - The limit in effect. `null` means this scope applies to the key but places no limit on it - spend: float | None - Spend as the enforcing check reads it, from the same cross-pod - counter, not the periodically-synced database column - - spend_state: str - Whether `spend` is `live`, missing because no counter exists yet - (`no_counter`), or missing because the read failed (`unavailable`) + counter, not the periodically-synced database column. `null` only when the read failed + - spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be + enforced because no counter has been created yet (`no_counter`), or is missing because the + read failed (`unavailable`) - remaining: float | None - `max_budget - spend`, when both are known - comparison: str - The operator the enforcing check uses, which differs per scope - budget_duration / budget_reset_at / window_start: When spend next resets to zero diff --git a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py index a843121f01c..d3d0ee7f752 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py @@ -17110,7 +17110,8 @@ async def test_key_budgets_probe_every_request_model_that_maps_onto_a_per_model_ assert all(row.entity_label == "gpt-5" and row.source == "key.model_max_budget[gpt-5]" for row in rows.values()) assert rows["openai/gpt-5"].spend == 6.0 assert rows["openai/gpt-5"].status == "exceeded" - assert rows["gpt-5"].spend is None + assert rows["gpt-5"].spend == 0.0 + assert rows["gpt-5"].spend_state == "no_counter" assert rows["gpt-5"].status == "ok", "a model with no counter of its own is not over the cap" @@ -17220,11 +17221,14 @@ async def test_key_budgets_warn_on_every_row_when_custom_auth_skips_the_read_tim @pytest.mark.asyncio -async def test_key_budgets_never_pair_a_missing_spend_state_with_a_number(): +async def test_key_budgets_only_blank_the_numbers_on_the_state_that_means_we_do_not_know(): """ - `spend_state` is the only thing that stops a blank cell rendering as `$0.00` at 0% of the meter, - so a row that says the number is missing must not also carry one, or anything derived from one. + The two non-live states are not the same kind of absence. A counter that does not exist yet is + enforced as zero, so the row carries that zero and the full headroom, and `spend_state` is what + says the zero is not a reading. A failed read knows nothing, so it carries nothing: sending zero + there would render as untouched headroom on a budget that may well be exhausted. """ + async def _explode(**kwargs): raise RuntimeError("redis is down") @@ -17238,8 +17242,12 @@ async def test_key_budgets_never_pair_a_missing_spend_state_with_a_number(): states = {e.spend_state for e in budgets} assert states == {"live", "no_counter", "unavailable"}, f"all three states must occur here, saw {states}" for entry in budgets: - assert (entry.spend is None) == (entry.spend_state != "live"), entry + assert (entry.spend is None) == (entry.spend_state == "unavailable"), entry assert entry.spend is not None or entry.remaining is None, entry + if entry.spend_state == "no_counter": + assert entry.spend == 0.0, entry + assert entry.remaining == entry.max_budget, entry + assert entry.status != "exceeded", entry @pytest.mark.asyncio @@ -17266,7 +17274,8 @@ async def test_key_budgets_tell_a_cold_per_model_counter_apart_from_a_budget_tha ) cold_entry = next(e for e in cold_budgets if e.scope == "key_model" and e.entity_id == "gpt-5") - assert cold_entry.spend is None + assert cold_entry.spend == 0.0, "the cap is compared against zero until a counter exists, so report zero" + assert cold_entry.remaining == cold_entry.max_budget assert cold_entry.spend_state == "no_counter" assert cold_entry.notes == warm_entry.notes, "a cold counter is the same budget, so it earns no extra caveat" assert _MODEL_BUDGET_NOTE in cold_entry.notes