mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(proxy)!: enforce fallback budget by default
A budget bypass that ships off by default stays open for every deployment that does not know to look for the flag, so `enforce_fallback_budget` now defaults to true and `general_settings.enforce_fallback_budget: false` is the opt-out for anyone who wants the old unguarded behaviour back. BREAKING CHANGE: a paid fallback target is now refused for callers who are over their key or user `max_budget`. Deployments relying on fallbacks to keep serving over-budget callers must set enforce_fallback_budget: false.
This commit is contained in:
parent
4a70bc3ba3
commit
cfe65f7b55
3 changed files with 26 additions and 7 deletions
|
|
@ -8,8 +8,8 @@ actually bills. So a free model with a paid fallback spends without a gate.
|
|||
|
||||
This predicate is injected into the router to re-check budget for each fallback target before it is
|
||||
attempted, mirroring `fallback_model_access.py`. It deliberately leaves the primary attempt alone:
|
||||
a zero-cost model is never blocked by budget, and only the paid fallback is refused. Opt-in via
|
||||
`general_settings.enforce_fallback_budget: true`.
|
||||
a zero-cost model is never blocked by budget, and only the paid fallback is refused. On by default;
|
||||
set `general_settings.enforce_fallback_budget: false` to restore the unguarded behaviour.
|
||||
|
||||
Scope: the key's and the user's `max_budget`. Not covered yet, and each needs a read-only evaluation
|
||||
path before it can be: team, team-member, end-user, org, global and per-model budgets, whose
|
||||
|
|
@ -50,7 +50,7 @@ class _RequestMetadata(BaseModel):
|
|||
|
||||
|
||||
class _FallbackBudgetSettings(BaseModel):
|
||||
enforce_fallback_budget: bool = False
|
||||
enforce_fallback_budget: bool = True
|
||||
|
||||
|
||||
def _token_in_metadata(metadata: object) -> UserAPIKeyAuth | None:
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from litellm.proxy._types import UserAPIKeyAuth
|
|||
from litellm.proxy.auth.fallback_budget import (
|
||||
RouterFallbackBudgetCheck,
|
||||
is_token_within_budget_for_model,
|
||||
router_fallback_budget_check,
|
||||
)
|
||||
|
||||
FREE_MODEL = {
|
||||
|
|
@ -182,3 +183,20 @@ async def test_router_without_a_budget_check_attempts_every_fallback():
|
|||
over = {"metadata": {"user_api_key_auth": _token(user_spend=1900.0, user_max_budget=50.0)}}
|
||||
|
||||
assert await _is_fallback_target_within_budget(router, "paid-model", "free-model", over) is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_enforcement_is_on_by_default_and_opt_out_restores_the_leak(monkeypatch):
|
||||
"""
|
||||
Leaving the paid fallback unguarded is the budget bypass this module exists to close, so an
|
||||
unconfigured proxy has to enforce. `enforce_fallback_budget: false` is the deliberate opt-out.
|
||||
"""
|
||||
from litellm.proxy import proxy_server
|
||||
|
||||
over = {"metadata": {"user_api_key_auth": _token(user_spend=1900.0, user_max_budget=50.0)}}
|
||||
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False)
|
||||
assert await router_fallback_budget_check(model="paid-model", request_kwargs=over, llm_router=_router()) is False
|
||||
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {"enforce_fallback_budget": False}, raising=False)
|
||||
assert await router_fallback_budget_check(model="paid-model", request_kwargs=over, llm_router=_router()) is True
|
||||
|
|
|
|||
|
|
@ -13416,14 +13416,15 @@ async def test_load_config_router_budget_checks_fallback_targets_against_the_cal
|
|||
}
|
||||
}
|
||||
|
||||
# off by default: the paid fallback is still attempted for an over-budget caller
|
||||
# on by default: an over-budget caller is refused the paid fallback with no config at all
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False)
|
||||
assert await router.fallback_budget_check(model="m", request_kwargs=over_budget, llm_router=router) is True
|
||||
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {"enforce_fallback_budget": True}, raising=False)
|
||||
assert await router.fallback_budget_check(model="m", request_kwargs=over_budget, llm_router=router) is False
|
||||
assert await router.fallback_budget_check(model="m", request_kwargs=under_budget, llm_router=router) is True
|
||||
|
||||
# explicit opt-out restores the unguarded behaviour
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {"enforce_fallback_budget": False}, raising=False)
|
||||
assert await router.fallback_budget_check(model="m", request_kwargs=over_budget, llm_router=router) is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_load_config_user_api_key_cache_max_size_keeps_more_than_200_entries(tmp_path, monkeypatch):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue