mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
test(e2e): add reproducer for unenforced end-user per-model rate limits
model_max_budget accepts an rpm_limit alongside the spend cap, and /budget/new
stores it: the create response echoes {"gemini-2.5-flash": {"rpm_limit": 1,
"max_budget": 100.0, "budget_duration": "1d"}}. Attach that budget to an end
user, drive three calls as that user, and all three return 200. The limit is
accepted, persisted, and then ignored.
The same shape already works when the budget hangs off a key, which is what
makes this quietly dangerous: the API gives every indication the cap is in
force. A customer using it to hold one end user to a slow rate on a shared key
gets no throttling at all.
Harness additions this needs: ModelBudgetEntry carries the rpm_limit/tpm_limit
the route already accepts, BudgetNewBody and create_budget carry
model_max_budget, and create_customer can attach an existing budget_id rather
than only an inline max_budget.
Red today, for the reason in the assertion message.
This commit is contained in:
parent
1eaf410989
commit
78a759d73c
4 changed files with 67 additions and 5 deletions
|
|
@ -18,6 +18,7 @@
|
|||
- {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"}
|
||||
- {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"}
|
||||
- {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"}
|
||||
- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"}
|
||||
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
|
||||
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
|
||||
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}
|
||||
|
|
|
|||
|
|
@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator
|
|||
class ModelBudgetEntry(BaseModel):
|
||||
budget_limit: float
|
||||
time_period: str
|
||||
rpm_limit: int | None = None
|
||||
tpm_limit: int | None = None
|
||||
|
||||
|
||||
class BudgetWindow(BaseModel):
|
||||
|
|
|
|||
|
|
@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel):
|
|||
|
||||
class CustomerNewBody(BaseModel):
|
||||
user_id: str
|
||||
max_budget: float
|
||||
max_budget: float | None = None
|
||||
budget_id: str | None = None
|
||||
|
||||
|
||||
class OrgNewBody(BaseModel):
|
||||
|
|
@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel):
|
|||
|
||||
|
||||
class BudgetNewBody(BaseModel):
|
||||
max_budget: float
|
||||
max_budget: float | None = None
|
||||
soft_budget: float | None = None
|
||||
budget_duration: str | None = None
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None
|
||||
|
||||
|
||||
class BudgetNewResponse(BaseModel):
|
||||
|
|
@ -326,11 +328,19 @@ class BudgetClient:
|
|||
|
||||
# ---- customer / end-user -------------------------------------------
|
||||
|
||||
def create_customer(self, customer_id: str, *, max_budget: float) -> str:
|
||||
def create_customer(
|
||||
self,
|
||||
customer_id: str,
|
||||
*,
|
||||
max_budget: float | None = None,
|
||||
budget_id: str | None = None,
|
||||
) -> str:
|
||||
resp = self.proxy.transport.send(
|
||||
"/customer/new",
|
||||
headers=self.proxy.transport.master,
|
||||
json=CustomerNewBody(user_id=customer_id, max_budget=max_budget),
|
||||
json=CustomerNewBody(
|
||||
user_id=customer_id, max_budget=max_budget, budget_id=budget_id
|
||||
),
|
||||
)
|
||||
assert resp.ok, resp.body
|
||||
return customer_id
|
||||
|
|
@ -509,9 +519,10 @@ class BudgetClient:
|
|||
def create_budget(
|
||||
self,
|
||||
*,
|
||||
max_budget: float,
|
||||
max_budget: float | None = None,
|
||||
soft_budget: float | None = None,
|
||||
budget_duration: str | None = None,
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None,
|
||||
) -> str:
|
||||
return unwrap(
|
||||
self.proxy.transport.post(
|
||||
|
|
@ -521,6 +532,7 @@ class BudgetClient:
|
|||
max_budget=max_budget,
|
||||
soft_budget=soft_budget,
|
||||
budget_duration=budget_duration,
|
||||
model_max_budget=model_max_budget,
|
||||
),
|
||||
response_type=BudgetNewResponse,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ import time
|
|||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block, model_budget
|
||||
from models import ModelBudgetEntry
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import require_successful_call
|
||||
from lifecycle import ResourceManager
|
||||
|
|
@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model(
|
|||
f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated"
|
||||
)
|
||||
require_successful_call(other)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit")
|
||||
def test_end_user_model_max_budget_enforces_per_model_rpm(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
"""A per-model rpm_limit on an end user's budget has to actually throttle.
|
||||
|
||||
`model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a
|
||||
customer uses it to hold one end user to a slow rate on an expensive model
|
||||
without limiting the shared key everyone else runs through. The budget is
|
||||
attached to the end user rather than to the key, which is the case that
|
||||
matters here: the same shape already works when the budget hangs off a key.
|
||||
"""
|
||||
budget_id = client.create_budget(
|
||||
model_max_budget={
|
||||
FREE_MODEL: ModelBudgetEntry(
|
||||
budget_limit=1000.0, time_period="1d", rpm_limit=1
|
||||
)
|
||||
}
|
||||
)
|
||||
resources.defer(lambda: client.delete_budget(budget_id))
|
||||
|
||||
customer = f"e2e-mmb-cust-{unique_marker()}"
|
||||
_ = client.create_customer(customer, budget_id=budget_id)
|
||||
resources.defer(lambda: client.delete_customers([customer]))
|
||||
|
||||
key = client.generate_key()
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
statuses = tuple(
|
||||
client.chat(
|
||||
key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer
|
||||
).status_code
|
||||
for _ in range(3)
|
||||
)
|
||||
|
||||
assert statuses[0] == 200, (
|
||||
f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}"
|
||||
)
|
||||
assert 429 in statuses[1:], (
|
||||
f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: "
|
||||
f"three calls returned {statuses}. The limit is accepted and stored by "
|
||||
f"/budget/new but never enforced for end-user budgets, so a customer "
|
||||
f"cannot rate-limit an individual end user on a shared key"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue