mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-07 08:26:10 +00:00
fix(budget): reject known estimates over remaining budget under fail_closed_budget_enforcement (#39214)
Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
346efa0c33
commit
3888a85045
2 changed files with 115 additions and 9 deletions
|
|
@ -120,13 +120,15 @@ async def _apply_over_budget_reservation_policy(
|
|||
applied_entries: list[dict[str, float | str]],
|
||||
reservation_cost: float,
|
||||
current_spend: float,
|
||||
fail_closed_budget_enforcement: bool = False,
|
||||
) -> float:
|
||||
"""
|
||||
Decide what to do when a counter is over budget, and return the reservation
|
||||
cost to carry into the next counter. Three outcomes: an over-budget key that
|
||||
opted into throttling releases its own reservation (the rate limiter slows
|
||||
it) and keeps the cost; a partially-remaining budget resizes the reservation
|
||||
down to what is left; anything else hard-blocks by raising.
|
||||
down to what is left, unless strict enforcement is on, because the known
|
||||
estimate already does not fit; anything else hard-blocks by raising.
|
||||
"""
|
||||
if _key_reservation_should_release_for_throttle(counter.counter_key, valid_token):
|
||||
await _release_applied_entries_best_effort(entries=[entry], default_reserved_cost=reservation_cost)
|
||||
|
|
@ -134,21 +136,36 @@ async def _apply_over_budget_reservation_policy(
|
|||
return reservation_cost
|
||||
|
||||
remaining_before_reservation: Final = counter.max_budget - (current_spend - reservation_cost)
|
||||
if remaining_before_reservation > 1e-12:
|
||||
await _resize_applied_reservation(
|
||||
entries=applied_entries,
|
||||
current_reserved_cost=reservation_cost,
|
||||
new_reserved_cost=remaining_before_reservation,
|
||||
if remaining_before_reservation <= 1e-12:
|
||||
_raise_counter_budget_exceeded(counter=counter, current_cost=current_spend)
|
||||
if fail_closed_budget_enforcement and current_spend - counter.max_budget > 1e-12:
|
||||
_raise_counter_budget_exceeded(
|
||||
counter=counter,
|
||||
current_cost=current_spend - reservation_cost,
|
||||
estimated_cost=reservation_cost,
|
||||
)
|
||||
return remaining_before_reservation
|
||||
await _resize_applied_reservation(
|
||||
entries=applied_entries,
|
||||
current_reserved_cost=reservation_cost,
|
||||
new_reserved_cost=remaining_before_reservation,
|
||||
)
|
||||
return remaining_before_reservation
|
||||
|
||||
|
||||
def _raise_counter_budget_exceeded(
|
||||
counter: _BudgetCounter,
|
||||
current_cost: float,
|
||||
estimated_cost: float | None = None,
|
||||
) -> NoReturn:
|
||||
estimate_detail: Final = "" if estimated_cost is None else f"Estimated request cost: {estimated_cost}, "
|
||||
raise litellm.BudgetExceededError(
|
||||
current_cost=current_spend,
|
||||
current_cost=current_cost,
|
||||
max_budget=counter.max_budget,
|
||||
message=(
|
||||
"Budget has been exceeded! "
|
||||
f"{counter.entity_type}={counter.entity_id} "
|
||||
f"Current cost: {current_spend}, "
|
||||
f"Current cost: {current_cost}, "
|
||||
f"{estimate_detail}"
|
||||
f"Max budget: {counter.max_budget}"
|
||||
),
|
||||
entity_type=_COUNTER_ENTITY_TYPES.get(counter.entity_type),
|
||||
|
|
@ -258,6 +275,7 @@ async def reserve_budget_for_request(
|
|||
applied_entries=applied_entries,
|
||||
reservation_cost=reservation_cost,
|
||||
current_spend=current_spend,
|
||||
fail_closed_budget_enforcement=fail_closed_budget_enforcement,
|
||||
)
|
||||
continue
|
||||
except Exception:
|
||||
|
|
|
|||
|
|
@ -819,6 +819,94 @@ async def test_should_cap_known_estimate_to_remaining_budget(
|
|||
) == pytest.approx(0.9)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fail_closed_rejects_known_estimate_exceeding_remaining_budget(
|
||||
spend_counter_state,
|
||||
):
|
||||
"""LIT-5922: with strict enforcement on, a request whose known estimate does
|
||||
not fit the remaining budget must be rejected before dispatch instead of
|
||||
having its reservation shrunk to the headroom and admitted, and the counter
|
||||
must be restored to the pre-request spend."""
|
||||
counter_cache, key_cache = spend_counter_state
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache)
|
||||
valid_token = UserAPIKeyAuth(
|
||||
token="key-budget-known-estimate-fail-closed",
|
||||
spend=0.9,
|
||||
max_budget=1.0,
|
||||
)
|
||||
counter_cache.in_memory_cache.set_cache(
|
||||
key="spend:key:key-budget-known-estimate-fail-closed",
|
||||
value=0.9,
|
||||
)
|
||||
|
||||
with patch( # test-quality-ok: reserve_budget_for_request takes no estimator, so pinning the estimate needs this attribute
|
||||
"litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost",
|
||||
return_value=0.6,
|
||||
):
|
||||
with pytest.raises(litellm.BudgetExceededError) as exc_info:
|
||||
await reserve_budget_for_request(
|
||||
request_body=_request_body(),
|
||||
route="/chat/completions",
|
||||
llm_router=None,
|
||||
valid_token=valid_token,
|
||||
team_object=None,
|
||||
user_object=None,
|
||||
prisma_client=None,
|
||||
user_api_key_cache=key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
fail_closed_budget_enforcement=True,
|
||||
)
|
||||
|
||||
assert exc_info.value.current_cost == pytest.approx(0.9)
|
||||
assert exc_info.value.max_budget == pytest.approx(1.0)
|
||||
assert "Current cost: 0.9, Estimated request cost: 0.6, Max budget: 1.0" in str(exc_info.value)
|
||||
assert counter_cache.in_memory_cache.get_cache(
|
||||
key="spend:key:key-budget-known-estimate-fail-closed"
|
||||
) == pytest.approx(0.9)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fail_closed_tolerates_float_noise_when_estimate_exactly_fits(
|
||||
spend_counter_state,
|
||||
):
|
||||
"""0.1 + 0.2 lands a hair above 0.3 in floating point. Strict enforcement
|
||||
must treat that as fitting the budget, not reject it."""
|
||||
counter_cache, key_cache = spend_counter_state
|
||||
proxy_logging_obj = ProxyLogging(user_api_key_cache=key_cache)
|
||||
valid_token = UserAPIKeyAuth(
|
||||
token="key-budget-fail-closed-float-noise",
|
||||
spend=0.1,
|
||||
max_budget=0.3,
|
||||
)
|
||||
counter_cache.in_memory_cache.set_cache(
|
||||
key="spend:key:key-budget-fail-closed-float-noise",
|
||||
value=0.1,
|
||||
)
|
||||
|
||||
with patch( # test-quality-ok: reserve_budget_for_request takes no estimator, so pinning the estimate needs this attribute
|
||||
"litellm.proxy.spend_tracking.budget_reservation.estimate_request_max_cost",
|
||||
return_value=0.2,
|
||||
):
|
||||
reservation = await reserve_budget_for_request(
|
||||
request_body=_request_body(),
|
||||
route="/chat/completions",
|
||||
llm_router=None,
|
||||
valid_token=valid_token,
|
||||
team_object=None,
|
||||
user_object=None,
|
||||
prisma_client=None,
|
||||
user_api_key_cache=key_cache,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
fail_closed_budget_enforcement=True,
|
||||
)
|
||||
|
||||
assert reservation is not None
|
||||
assert reservation["reserved_cost"] == pytest.approx(0.2)
|
||||
assert counter_cache.in_memory_cache.get_cache(
|
||||
key="spend:key:key-budget-fail-closed-float-noise"
|
||||
) == pytest.approx(0.3)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_should_clamp_reservation_to_default_when_output_cap_missing(
|
||||
spend_counter_state,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue