From b49c4befe9ee530f8e06e958fff135a70a0d9d85 Mon Sep 17 00:00:00 2001 From: willcai1984 Date: Fri, 28 Aug 2026 16:04:34 +0800 Subject: [PATCH] feat(proxy): opt-in per-request budget reservation cost cap Adds LITELLM_BUDGET_RESERVATION_MAX_COST_USD (disabled unless set to a positive finite number) to clamp the worst-case budget reservation of a single request. Clients with large contexts and high max_tokens (coding agents) produce per-request estimates of several dollars, so a moderate concurrent burst pre-fills the admission counter to exactly max_budget and 429s every request while recorded spend stays near zero. The cap is documented as an availability trade-off, not a strict budget bound: worst-case transient overshoot is sum(max(actual_cost - cap, 0)) over capped in-flight requests. Because a capped reservation under-reserves, the cancel path must not clamp the recorded input cost to the reservation: input_cost now keeps the true estimated input cost so reconcile lands the already-billed amount (success reconcile was already exact). Tests cover: default off, cap limits reservation + cancel reconciles true input cost (idempotent), <=0 disables, invalid/NaN/inf disable with strict reservations, success reconcile records true cost. --- .../spend_tracking/budget_reservation.py | 87 ++++++++- .../proxy/test_budget_reservation.py | 181 ++++++++++++++++++ 2 files changed, 262 insertions(+), 6 deletions(-) diff --git a/litellm/proxy/spend_tracking/budget_reservation.py b/litellm/proxy/spend_tracking/budget_reservation.py index e28fa2c06a4..063fe31311b 100644 --- a/litellm/proxy/spend_tracking/budget_reservation.py +++ b/litellm/proxy/spend_tracking/budget_reservation.py @@ -3,10 +3,12 @@ from __future__ import annotations import asyncio import json import math +import os import time from collections.abc import Mapping, Sequence from dataclasses import dataclass from datetime import datetime, timedelta, timezone +from functools import lru_cache from types import MappingProxyType from typing import Final, NoReturn, SupportsFloat, SupportsIndex, SupportsInt, cast @@ -291,11 +293,13 @@ async def reserve_budget_for_request( ) current_spend_by_counter_key: Final[dict[str, float]] = {} - reservation_cost = estimate_request_max_cost( - request_body=request_body, - route=route, - llm_router=llm_router, - input_token_counts=input_token_counts, + reservation_cost = cap_reservation_cost( # rebind-ok: admission policy may resize to remaining budget + reservation_cost=estimate_request_max_cost( + request_body=request_body, + route=route, + llm_router=llm_router, + input_token_counts=input_token_counts, + ), ) # estimate_request_max_cost still returns None when the model is unknown # to the cost map (no token-priced cost fields, e.g. image/audio routes). @@ -367,7 +371,13 @@ async def reserve_budget_for_request( "entries": applied_entries, "finalized": False, "callback_bound": False, - "input_cost": min(float(input_cost or 0.0), reservation_cost), + # True input-token cost, deliberately NOT clamped to the (possibly + # capped) reservation: cancel-path reconcile lands this cost on the + # counter via the actual - reserved delta, so clamping it would + # under-record a cancelled request below what the provider already + # billed whenever the true input cost exceeds the cap (very large + # contexts). reserved_cost stays the capped pre-occupation. + "input_cost": float(input_cost or 0.0), "input_tokens": max(input_token_counts.values(), default=None), } _start_reservation_lease_renewal( @@ -1187,6 +1197,71 @@ def _coerce_datetime(value: object) -> datetime | None: return None +@lru_cache(maxsize=1) +def reservation_cost_cap() -> float | None: + """Opt-in per-request budget-reservation cap, in USD. + + Parsed once per worker (invalid-value warnings are emitted once at startup/ + first use, not once per request). Disabled unless + ``LITELLM_BUDGET_RESERVATION_MAX_COST_USD`` is set to a + positive, finite number; unset values and values <= 0 both disable the + cap, and non-finite values are treated as unset (with a warning) rather + than poisoning the spend counters. + + ⚠️ The cap is an operations availability trade-off, NOT a strict budget + upper bound: clients with large contexts and high max_tokens (coding + agents) yield worst-case estimates of several dollars, so a moderately + concurrent burst can pre-fill the counter to exactly max_budget and 429 + every request while recorded spend stays near zero. A capped reservation + under-reserves, so the worst-case concurrent overshoot against a budget + is the sum of max(actual_cost - cap, 0) over capped in-flight requests. + Post-call reconcile still records true spend, so enforcement resumes as + soon as the counter reflects it — but a burst can transiently exceed the + budget by that amount. Strict per-request bounds require leaving the cap + disabled. + """ + raw: Final = os.getenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD") + if raw is None or not raw.strip(): + return None + try: + value: Final = float(raw) + except ValueError: + verbose_proxy_logger.warning( + "Invalid LITELLM_BUDGET_RESERVATION_MAX_COST_USD=%r (must be a number); disabling the reservation cap", + raw, + ) + return None + if not math.isfinite(value): + verbose_proxy_logger.warning( + "Invalid LITELLM_BUDGET_RESERVATION_MAX_COST_USD=%r (must be finite); disabling the reservation cap", + raw, + ) + return None + if value <= 0: + return None + return value + + +def cap_reservation_cost(reservation_cost: float | None) -> float | None: + """Clamp a worst-case reservation estimate to the configured cap, if any. + + ``None`` (unknown-cost route) passes through unchanged so the caller keeps + its read-time-enforcement fallback. The cap only shrinks the in-flight + pre-occupation; post-call reconcile records the request's true cost. + """ + if reservation_cost is None: + return None + cap: Final = reservation_cost_cap() + if cap is None or reservation_cost <= cap: + return reservation_cost + verbose_proxy_logger.debug( + "Budget reservation capped: estimate=%.4f -> cap=%.4f", + reservation_cost, + cap, + ) + return cap + + def estimate_request_max_cost( request_body: dict, route: str, diff --git a/tests/test_litellm/proxy/test_budget_reservation.py b/tests/test_litellm/proxy/test_budget_reservation.py index 18b046cd83c..9869a280b52 100644 --- a/tests/test_litellm/proxy/test_budget_reservation.py +++ b/tests/test_litellm/proxy/test_budget_reservation.py @@ -39,11 +39,13 @@ from litellm.proxy.common_utils.user_api_key_cache import ( model_access_group_spend_counter_key, ) from litellm.proxy.spend_tracking.budget_reservation import ( + reservation_cost_cap, _get_model_access_group_budget_counters, estimate_request_max_cost, get_budget_window_start, invalidate_budget_reservation_counters, release_budget_reservation, + reconcile_budget_reservation, release_budget_reservation_on_cancel, reserve_budget_for_request, ) @@ -3669,3 +3671,182 @@ async def test_unreserved_model_access_group_is_charged_alongside_a_reserved_one assert counter_cache.in_memory_cache.get_cache( key=model_access_group_spend_counter_key("starter") ) == pytest.approx(4.2) + + +@pytest.mark.asyncio +async def test_reservation_cost_cap_disabled_by_default(spend_counter_state, monkeypatch): + """Without the env override the reservation keeps its strict worst-case size.""" + monkeypatch.delenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD", raising=False) + reservation_cost_cap.cache_clear() + counter_cache, key_cache = spend_counter_state + proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache) + valid_token = UserAPIKeyAuth(token="key-cap-disabled", spend=0.0, max_budget=10.0) + + with patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost", + return_value=3.0, + ): + reservation = await reserve_budget_for_request( + request_body=_request_body(), + route="/chat/completions", + llm_router=None, + valid_token=valid_token, + team_object=None, + user_object=None, + prisma_client=None, + user_api_key_cache=key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + + assert reservation is not None + assert reservation["reserved_cost"] == pytest.approx(3.0) + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-disabled") == pytest.approx(3.0) + await release_budget_reservation(reservation) + + +@pytest.mark.asyncio +async def test_cost_cap_limits_reservation_and_cancel_reconciles_true_input_cost( + spend_counter_state, + monkeypatch, +): + """With a cap, the in-flight pre-occupation shrinks but a cancelled request + still reconciles the counter to its TRUE input cost, not to the capped + reservation — the provider already billed those input tokens.""" + monkeypatch.setenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD", "0.1") + reservation_cost_cap.cache_clear() + counter_cache, key_cache = spend_counter_state + proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache) + valid_token = UserAPIKeyAuth(token="key-cap-cancel", spend=0.0, max_budget=10.0) + + with ( + patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost", + return_value=3.0, + ), + patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_input_cost", + return_value=0.5, + ), + ): + reservation = await reserve_budget_for_request( + request_body=_request_body(), + route="/chat/completions", + llm_router=None, + valid_token=valid_token, + team_object=None, + user_object=None, + prisma_client=None, + user_api_key_cache=key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + + assert reservation is not None + assert reservation["reserved_cost"] == pytest.approx(0.1) + assert reservation["input_cost"] == pytest.approx(0.5) + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-cancel") == pytest.approx(0.1) + + await release_budget_reservation_on_cancel(reservation) + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-cancel") == pytest.approx(0.5) + assert reservation["finalized"] is True + + # idempotent: a second cancel reconcile must not change the counter again + await release_budget_reservation_on_cancel(reservation) + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-cancel") == pytest.approx(0.5) + + +@pytest.mark.asyncio +async def test_cost_cap_zero_disables_cap(spend_counter_state, monkeypatch): + monkeypatch.setenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD", "0") + reservation_cost_cap.cache_clear() + counter_cache, key_cache = spend_counter_state + proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache) + valid_token = UserAPIKeyAuth(token="key-cap-zero", spend=0.0, max_budget=10.0) + + with patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost", + return_value=3.0, + ): + reservation = await reserve_budget_for_request( + request_body=_request_body(), + route="/chat/completions", + llm_router=None, + valid_token=valid_token, + team_object=None, + user_object=None, + prisma_client=None, + user_api_key_cache=key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + + assert reservation is not None + assert reservation["reserved_cost"] == pytest.approx(3.0) + await release_budget_reservation(reservation) + + +@pytest.mark.parametrize("raw_value", ["abc", "nan", "inf", "-inf"]) +@pytest.mark.asyncio +async def test_cost_cap_non_finite_or_invalid_values_disable_cap( + spend_counter_state, + monkeypatch, + raw_value, +): + """Non-numeric and non-finite env values must not poison the counters; + they disable the cap (strict reservations) instead.""" + monkeypatch.setenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD", raw_value) + reservation_cost_cap.cache_clear() + counter_cache, key_cache = spend_counter_state + proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache) + valid_token = UserAPIKeyAuth(token="key-cap-invalid", spend=0.0, max_budget=10.0) + + with patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost", + return_value=3.0, + ): + reservation = await reserve_budget_for_request( + request_body=_request_body(), + route="/chat/completions", + llm_router=None, + valid_token=valid_token, + team_object=None, + user_object=None, + prisma_client=None, + user_api_key_cache=key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + + assert reservation is not None + assert reservation["reserved_cost"] == pytest.approx(3.0) + await release_budget_reservation(reservation) + + +@pytest.mark.asyncio +async def test_cost_cap_success_reconcile_records_true_cost(spend_counter_state, monkeypatch): + """Under a cap, the success reconcile must land the request's TRUE cost on + the counter, even though the in-flight reservation only pre-occupied the + capped amount.""" + monkeypatch.setenv("LITELLM_BUDGET_RESERVATION_MAX_COST_USD", "0.1") + reservation_cost_cap.cache_clear() + counter_cache, key_cache = spend_counter_state + proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache) + valid_token = UserAPIKeyAuth(token="key-cap-success", spend=0.0, max_budget=10.0) + + with patch( # test-quality-ok: isolate estimator output to exercise reservation-state transitions + "litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost", + return_value=3.0, + ): + reservation = await reserve_budget_for_request( + request_body=_request_body(), + route="/chat/completions", + llm_router=None, + valid_token=valid_token, + team_object=None, + user_object=None, + prisma_client=None, + user_api_key_cache=key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + assert reservation is not None + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-success") == pytest.approx(0.1) + + await reconcile_budget_reservation(budget_reservation=reservation, actual_cost=0.6) + assert counter_cache.in_memory_cache.get_cache(key="spend:key:key-cap-success") == pytest.approx(0.6)