From 78a759d73c6b11340a540d2ffa585b821670c927 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Sat, 25 Jul 2026 14:10:03 -0700 Subject: [PATCH] test(e2e): add reproducer for unenforced end-user per-model rate limits model_max_budget accepts an rpm_limit alongside the spend cap, and /budget/new stores it: the create response echoes {"gemini-2.5-flash": {"rpm_limit": 1, "max_budget": 100.0, "budget_duration": "1d"}}. Attach that budget to an end user, drive three calls as that user, and all three return 200. The limit is accepted, persisted, and then ignored. The same shape already works when the budget hangs off a key, which is what makes this quietly dangerous: the API gives every indication the cap is in force. A customer using it to hold one end user to a slow rate on a shared key gets no throttling at all. Harness additions this needs: ModelBudgetEntry carries the rpm_limit/tpm_limit the route already accepts, BudgetNewBody and create_budget carry model_max_budget, and create_customer can attach an existing budget_id rather than only an inline max_budget. Red today, for the reason in the assertion message. --- .../coverage_registry/quota_management.yaml | 1 + tests/e2e/models.py | 2 + .../quota_management/budgets/budget_client.py | 22 +++++++-- .../budgets/test_model_max_budget_e2e.py | 47 +++++++++++++++++++ 4 files changed, 67 insertions(+), 5 deletions(-) diff --git a/tests/e2e/coverage_registry/quota_management.yaml b/tests/e2e/coverage_registry/quota_management.yaml index a8d0749cd8d..e853e631cee 100644 --- a/tests/e2e/coverage_registry/quota_management.yaml +++ b/tests/e2e/coverage_registry/quota_management.yaml @@ -18,6 +18,7 @@ - {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"} - {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"} - {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"} +- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"} - {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"} - {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"} - {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"} diff --git a/tests/e2e/models.py b/tests/e2e/models.py index af695acaa5e..a4e90c27f50 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator class ModelBudgetEntry(BaseModel): budget_limit: float time_period: str + rpm_limit: int | None = None + tpm_limit: int | None = None class BudgetWindow(BaseModel): diff --git a/tests/e2e/quota_management/budgets/budget_client.py b/tests/e2e/quota_management/budgets/budget_client.py index 83e8f27b597..5b9253928af 100644 --- a/tests/e2e/quota_management/budgets/budget_client.py +++ b/tests/e2e/quota_management/budgets/budget_client.py @@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel): class CustomerNewBody(BaseModel): user_id: str - max_budget: float + max_budget: float | None = None + budget_id: str | None = None class OrgNewBody(BaseModel): @@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel): class BudgetNewBody(BaseModel): - max_budget: float + max_budget: float | None = None soft_budget: float | None = None budget_duration: str | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetNewResponse(BaseModel): @@ -326,11 +328,19 @@ class BudgetClient: # ---- customer / end-user ------------------------------------------- - def create_customer(self, customer_id: str, *, max_budget: float) -> str: + def create_customer( + self, + customer_id: str, + *, + max_budget: float | None = None, + budget_id: str | None = None, + ) -> str: resp = self.proxy.transport.send( "/customer/new", headers=self.proxy.transport.master, - json=CustomerNewBody(user_id=customer_id, max_budget=max_budget), + json=CustomerNewBody( + user_id=customer_id, max_budget=max_budget, budget_id=budget_id + ), ) assert resp.ok, resp.body return customer_id @@ -509,9 +519,10 @@ class BudgetClient: def create_budget( self, *, - max_budget: float, + max_budget: float | None = None, soft_budget: float | None = None, budget_duration: str | None = None, + model_max_budget: dict[str, ModelBudgetEntry] | None = None, ) -> str: return unwrap( self.proxy.transport.post( @@ -521,6 +532,7 @@ class BudgetClient: max_budget=max_budget, soft_budget=soft_budget, budget_duration=budget_duration, + model_max_budget=model_max_budget, ), response_type=BudgetNewResponse, ) diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index 4d0df2c35ea..eb08da62d6b 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -11,6 +11,7 @@ import time import pytest from budget_client import BudgetClient, is_budget_block, model_budget +from models import ModelBudgetEntry from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager @@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model( f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" ) require_successful_call(other) + + +@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit") +def test_end_user_model_max_budget_enforces_per_model_rpm( + client: BudgetClient, resources: ResourceManager +) -> None: + """A per-model rpm_limit on an end user's budget has to actually throttle. + + `model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a + customer uses it to hold one end user to a slow rate on an expensive model + without limiting the shared key everyone else runs through. The budget is + attached to the end user rather than to the key, which is the case that + matters here: the same shape already works when the budget hangs off a key. + """ + budget_id = client.create_budget( + model_max_budget={ + FREE_MODEL: ModelBudgetEntry( + budget_limit=1000.0, time_period="1d", rpm_limit=1 + ) + } + ) + resources.defer(lambda: client.delete_budget(budget_id)) + + customer = f"e2e-mmb-cust-{unique_marker()}" + _ = client.create_customer(customer, budget_id=budget_id) + resources.defer(lambda: client.delete_customers([customer])) + + key = client.generate_key() + resources.defer(lambda: client.delete_key(key)) + + statuses = tuple( + client.chat( + key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer + ).status_code + for _ in range(3) + ) + + assert statuses[0] == 200, ( + f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}" + ) + assert 429 in statuses[1:], ( + f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: " + f"three calls returned {statuses}. The limit is accepted and stored by " + f"/budget/new but never enforced for end-user budgets, so a customer " + f"cannot rate-limit an individual end user on a shared key" + )