test(e2e): add reproducer for unenforced end-user per-model rate limits

model_max_budget accepts an rpm_limit alongside the spend cap, and /budget/new
stores it: the create response echoes {"gemini-2.5-flash": {"rpm_limit": 1,
"max_budget": 100.0, "budget_duration": "1d"}}. Attach that budget to an end
user, drive three calls as that user, and all three return 200. The limit is
accepted, persisted, and then ignored.

The same shape already works when the budget hangs off a key, which is what
makes this quietly dangerous: the API gives every indication the cap is in
force. A customer using it to hold one end user to a slow rate on a shared key
gets no throttling at all.

Harness additions this needs: ModelBudgetEntry carries the rpm_limit/tpm_limit
the route already accepts, BudgetNewBody and create_budget carry
model_max_budget, and create_customer can attach an existing budget_id rather
than only an inline max_budget.

Red today, for the reason in the assertion message.
This commit is contained in:
mubashir1osmani 2026-07-25 14:10:03 -07:00
parent 1eaf410989
commit 78a759d73c
4 changed files with 67 additions and 5 deletions

View file

@ -18,6 +18,7 @@
- {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"}
- {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"}
- {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"}
- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"}
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}

View file

@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator
class ModelBudgetEntry(BaseModel):
budget_limit: float
time_period: str
rpm_limit: int | None = None
tpm_limit: int | None = None
class BudgetWindow(BaseModel):

View file

@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel):
class CustomerNewBody(BaseModel):
user_id: str
max_budget: float
max_budget: float | None = None
budget_id: str | None = None
class OrgNewBody(BaseModel):
@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel):
class BudgetNewBody(BaseModel):
max_budget: float
max_budget: float | None = None
soft_budget: float | None = None
budget_duration: str | None = None
model_max_budget: dict[str, ModelBudgetEntry] | None = None
class BudgetNewResponse(BaseModel):
@ -326,11 +328,19 @@ class BudgetClient:
# ---- customer / end-user -------------------------------------------
def create_customer(self, customer_id: str, *, max_budget: float) -> str:
def create_customer(
self,
customer_id: str,
*,
max_budget: float | None = None,
budget_id: str | None = None,
) -> str:
resp = self.proxy.transport.send(
"/customer/new",
headers=self.proxy.transport.master,
json=CustomerNewBody(user_id=customer_id, max_budget=max_budget),
json=CustomerNewBody(
user_id=customer_id, max_budget=max_budget, budget_id=budget_id
),
)
assert resp.ok, resp.body
return customer_id
@ -509,9 +519,10 @@ class BudgetClient:
def create_budget(
self,
*,
max_budget: float,
max_budget: float | None = None,
soft_budget: float | None = None,
budget_duration: str | None = None,
model_max_budget: dict[str, ModelBudgetEntry] | None = None,
) -> str:
return unwrap(
self.proxy.transport.post(
@ -521,6 +532,7 @@ class BudgetClient:
max_budget=max_budget,
soft_budget=soft_budget,
budget_duration=budget_duration,
model_max_budget=model_max_budget,
),
response_type=BudgetNewResponse,
)

View file

@ -11,6 +11,7 @@ import time
import pytest
from budget_client import BudgetClient, is_budget_block, model_budget
from models import ModelBudgetEntry
from e2e_config import unique_marker
from e2e_http import require_successful_call
from lifecycle import ResourceManager
@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model(
f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated"
)
require_successful_call(other)
@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit")
def test_end_user_model_max_budget_enforces_per_model_rpm(
client: BudgetClient, resources: ResourceManager
) -> None:
"""A per-model rpm_limit on an end user's budget has to actually throttle.
`model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a
customer uses it to hold one end user to a slow rate on an expensive model
without limiting the shared key everyone else runs through. The budget is
attached to the end user rather than to the key, which is the case that
matters here: the same shape already works when the budget hangs off a key.
"""
budget_id = client.create_budget(
model_max_budget={
FREE_MODEL: ModelBudgetEntry(
budget_limit=1000.0, time_period="1d", rpm_limit=1
)
}
)
resources.defer(lambda: client.delete_budget(budget_id))
customer = f"e2e-mmb-cust-{unique_marker()}"
_ = client.create_customer(customer, budget_id=budget_id)
resources.defer(lambda: client.delete_customers([customer]))
key = client.generate_key()
resources.defer(lambda: client.delete_key(key))
statuses = tuple(
client.chat(
key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer
).status_code
for _ in range(3)
)
assert statuses[0] == 200, (
f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}"
)
assert 429 in statuses[1:], (
f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: "
f"three calls returned {statuses}. The limit is accepted and stored by "
f"/budget/new but never enforced for end-user budgets, so a customer "
f"cannot rate-limit an individual end user on a shared key"
)