mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
* test(e2e): budget reset diagonal for team, org, user, and #32005 team-member keys Adds E2E-7/8/10/11 from the budget-level x key-kind coverage matrix: each budget level serves traffic again after its budget_duration window elapses, walking the same ladder as the enforcement diagonal. New registry rows and tests cover the team, organization, and internal-user reset rungs, plus the #32005 interplay where a team-member key frozen by its owner's user budget comes back when the user's window renews; the bare-key and per-team-member rungs already had coverage Each case isolates the cap to one entity, drives spend to a budget_exceeded block, then polls past the window until a call succeeds, holding every refusal as a budget block so a reset that no-ops (stays blocked forever) or crashes (leaks a 5xx) fails the test. budget_duration becomes an optional param on the budget_client create_team / create_user / create_org helpers * test(e2e): fold the reset diagonal into test_budget_reset_e2e.py and address greptile nits Move the team / org / user / #32005 reset cases out of the standalone test_budget_reset_diagonal_e2e.py and into test_budget_reset_e2e.py, absorbing the pre-existing bare-key reset into the same TestBudgetResetDiagonal spec class so the whole reset ladder reads as one file (mirroring how the enforcement diagonal lives in test_budget_enforcement_e2e.py) and the drive/poll helpers are defined once instead of duplicated across reset files. Greptile nits: bound the drive phase to under one window (12 attempts x 2s < 30s) so a block is observed before the reset job can fire, and replace the bare assert in the poll loop with a pytest.fail that prints the HTTP status, so a provider 429 or a crashed reset path is distinguishable from a budget block at a glance. * test(e2e): trim reset diagonal docstrings back to the file's original style * test(e2e): inline single-use drive-loop bounds * test(e2e): cut the reset module docstring to one line * test(e2e): make the org reset test wait for a scheduled window (bugbot) /organization/new stores budget_duration without scheduling budget_reset_at, so the reset job's NULL catch-up branch zeroes org spend on its first 5-10s tick; the org reset test could pass off that catch-up instead of a real window roll (tracked as LIT-4570). The test now reads the org's budget_id and polls /budget/info until budget_reset_at is scheduled before driving spend, so the recovery it observes can only come from a genuine window expiry. Verified live: the org case now runs ~33s (a full window) instead of beating the rescheduler
This commit is contained in:
parent
9b0a424000
commit
dbb5b813c1
3 changed files with 132 additions and 39 deletions
|
|
@ -18,6 +18,9 @@
|
|||
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
|
||||
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
|
||||
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}
|
||||
- {id: quota_management.budget.team.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: team, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes a team's spend after the window; every key on the team serves again"}
|
||||
- {id: quota_management.budget.organization.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: organization, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "An org budget resets after its window; keys under the org serve again"}
|
||||
- {id: quota_management.budget.internal_user.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: internal_user, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "An internal user's budget resets after its window; their personal and team-member keys serve again"}
|
||||
- {id: quota_management.budget.team_member.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Member per-team budget reset keeps advancing window after window"}
|
||||
- {id: quota_management.budget.key_multi_window.blocks_then_resets, module: quota_management, tier: P1, behavior: budget, variant: key_multi_window, assertions: [blocks_then_resets], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_limits enforce within a short window and serve again in the next"}
|
||||
- {id: quota_management.budget.key_multi_window.resets_windows_independently, module: quota_management, tier: P2, behavior: budget, variant: key_multi_window, assertions: [resets_windows_independently], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Each window of a multi-window budget resets on its own schedule"}
|
||||
|
|
|
|||
|
|
@ -33,6 +33,7 @@ _TEAM_READY_SLEEP_SECONDS = 0.4
|
|||
|
||||
class UserNewBody(BaseModel):
|
||||
max_budget: float
|
||||
budget_duration: str | None = None
|
||||
|
||||
|
||||
class UserNewResponse(BaseModel):
|
||||
|
|
@ -64,6 +65,7 @@ class CustomerNewBody(BaseModel):
|
|||
class OrgNewBody(BaseModel):
|
||||
organization_alias: str
|
||||
max_budget: float
|
||||
budget_duration: str | None = None
|
||||
|
||||
|
||||
class OrgNewResponse(BaseModel):
|
||||
|
|
@ -74,6 +76,14 @@ class OrgDeleteBody(BaseModel):
|
|||
organization_ids: list[str]
|
||||
|
||||
|
||||
class OrgInfoParams(BaseModel):
|
||||
organization_id: str
|
||||
|
||||
|
||||
class OrgInfoResponse(BaseModel):
|
||||
budget_id: str | None = None
|
||||
|
||||
|
||||
class TeamMember(BaseModel):
|
||||
role: str
|
||||
user_id: str
|
||||
|
|
@ -82,6 +92,7 @@ class TeamMember(BaseModel):
|
|||
class TeamNewBody(BaseModel):
|
||||
team_alias: str
|
||||
max_budget: float | None = None
|
||||
budget_duration: str | None = None
|
||||
organization_id: str | None = None
|
||||
budget_limits: list[BudgetWindow] | None = None
|
||||
|
||||
|
|
@ -257,12 +268,12 @@ class BudgetClient:
|
|||
|
||||
# ---- internal user --------------------------------------------------
|
||||
|
||||
def create_user(self, *, max_budget: float) -> str:
|
||||
def create_user(self, *, max_budget: float, budget_duration: str | None = None) -> str:
|
||||
return unwrap(
|
||||
self.gateway.transport.post(
|
||||
"/user/new",
|
||||
headers=self.gateway.transport.master,
|
||||
json=UserNewBody(max_budget=max_budget),
|
||||
json=UserNewBody(max_budget=max_budget, budget_duration=budget_duration),
|
||||
response_type=UserNewResponse,
|
||||
)
|
||||
).user_id
|
||||
|
|
@ -301,16 +312,36 @@ class BudgetClient:
|
|||
|
||||
# ---- organization ---------------------------------------------------
|
||||
|
||||
def create_org(self, *, max_budget: float, alias: str) -> str:
|
||||
def create_org(self, *, max_budget: float, alias: str, budget_duration: str | None = None) -> str:
|
||||
return unwrap(
|
||||
self.gateway.transport.post(
|
||||
"/organization/new",
|
||||
headers=self.gateway.transport.master,
|
||||
json=OrgNewBody(organization_alias=alias, max_budget=max_budget),
|
||||
json=OrgNewBody(
|
||||
organization_alias=alias,
|
||||
max_budget=max_budget,
|
||||
budget_duration=budget_duration,
|
||||
),
|
||||
response_type=OrgNewResponse,
|
||||
)
|
||||
).organization_id
|
||||
|
||||
def org_budget_id(self, org_id: str) -> str | None:
|
||||
"""The id of the budget row backing an org; its budget_reset_at is read via
|
||||
budget_info (LIT-4570: /organization/new stores budget_duration without
|
||||
scheduling budget_reset_at, so the reset job's first tick schedules it)."""
|
||||
result = self.gateway.transport.get(
|
||||
"/organization/info",
|
||||
headers=self.gateway.transport.master,
|
||||
params=OrgInfoParams(organization_id=org_id),
|
||||
response_type=OrgInfoResponse,
|
||||
)
|
||||
match result:
|
||||
case Success(data=data):
|
||||
return data.budget_id
|
||||
case _:
|
||||
return None
|
||||
|
||||
def delete_org(self, org_id: str) -> None:
|
||||
_ = self.gateway.transport.delete(
|
||||
"/organization/delete",
|
||||
|
|
@ -326,6 +357,7 @@ class BudgetClient:
|
|||
*,
|
||||
alias: str,
|
||||
max_budget: float | None = None,
|
||||
budget_duration: str | None = None,
|
||||
organization_id: str | None = None,
|
||||
budget_limits: list[BudgetWindow] | None = None,
|
||||
) -> str:
|
||||
|
|
@ -336,6 +368,7 @@ class BudgetClient:
|
|||
json=TeamNewBody(
|
||||
team_alias=alias,
|
||||
max_budget=max_budget,
|
||||
budget_duration=budget_duration,
|
||||
organization_id=organization_id,
|
||||
budget_limits=budget_limits,
|
||||
),
|
||||
|
|
|
|||
|
|
@ -1,12 +1,4 @@
|
|||
"""Live e2e: a key budget resets (zeroes spend) after its budget_duration.
|
||||
|
||||
Short budget_duration (30s) + the fast-rescheduled reset job: a key blocked for
|
||||
exceeding its max_budget starts succeeding again once the duration elapses and the
|
||||
reset job zeroes key.spend. Closes the reset-zeroing gap in
|
||||
BUDGET_TEST_COVERAGE_MATRIX.md (reset_budget_for_litellm_keys), which the unit
|
||||
suite covers but no live test did - distinct from the per-window reset in
|
||||
test_multi_window_budget_e2e.py.
|
||||
"""
|
||||
"""Live e2e: an entity blocked over its max_budget serves again after its budget_duration window."""
|
||||
|
||||
import time
|
||||
|
||||
|
|
@ -19,42 +11,107 @@ from lifecycle import ResourceManager
|
|||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
TINY_CAP = 3e-6
|
||||
WINDOW = "30s"
|
||||
RESET_DEADLINE_SECONDS = 150
|
||||
|
||||
|
||||
def _call(client: BudgetClient, key: str):
|
||||
return client.chat(
|
||||
key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16
|
||||
)
|
||||
return client.chat(key, "claude-haiku-4-5", f"reset {unique_marker()}", max_tokens=16)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
|
||||
def test_key_budget_resets_after_duration(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = client.generate_key(max_budget=3e-6, budget_duration="30s")
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
# 1. exceed the budget -> litellm returns budget_exceeded
|
||||
blocked = False
|
||||
for _ in range(20):
|
||||
def _drive_to_block(client: BudgetClient, key: str) -> None:
|
||||
"""Spend until the cap blocks a call, staying under one window so the block
|
||||
is observed before the reset job can fire; fail hard if enforcement never trips."""
|
||||
for _ in range(12):
|
||||
result = _call(client, key)
|
||||
if is_budget_block(result):
|
||||
blocked = True
|
||||
break
|
||||
return
|
||||
require_successful_call(result)
|
||||
time.sleep(2)
|
||||
assert blocked, "key budget never enforced"
|
||||
pytest.fail("budget never enforced before the window could reset")
|
||||
|
||||
# 2. once the 30s duration elapses + the reset job runs, key.spend zeroes and
|
||||
# calls flow again. The window is wall-clock-aligned, so the reset lands up to
|
||||
# a window later, then the rescheduler (~15-20s) zeroes the spend; allow
|
||||
# generous headroom over that. A stuck rescheduler is caught by the wait-loop
|
||||
# timeout, not this elapsed bound.
|
||||
start = time.monotonic()
|
||||
while time.monotonic() < start + 150:
|
||||
|
||||
def _poll_until_serves_again(client: BudgetClient, key: str) -> None:
|
||||
"""Poll past the window until the blocked key serves again; every refusal must
|
||||
stay a budget block, so a crashed reset path or provider error fails loudly."""
|
||||
deadline = time.monotonic() + RESET_DEADLINE_SECONDS
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(5)
|
||||
result = _call(client, key)
|
||||
if result.ok:
|
||||
assert time.monotonic() - start < 120, "reset too slow for a 30s budget"
|
||||
return
|
||||
assert is_budget_block(result), f"non-budget error: {result.body[:200]}"
|
||||
pytest.fail("key budget never reset within 150s")
|
||||
if not is_budget_block(result):
|
||||
pytest.fail(f"non-budget error during reset wait: HTTP {result.status_code}: {result.body[:200]}")
|
||||
pytest.fail(f"budget never reset within {RESET_DEADLINE_SECONDS}s")
|
||||
|
||||
|
||||
class TestBudgetResetDiagonal:
|
||||
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
|
||||
def test_bare_key_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
|
||||
key = client.generate_key(max_budget=TINY_CAP, budget_duration=WINDOW)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_drive_to_block(client, key)
|
||||
_poll_until_serves_again(client, key)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team.resets_after_window")
|
||||
def test_team_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
|
||||
team_id = client.create_team(
|
||||
alias=f"e2e-team-reset-{unique_marker()}", max_budget=TINY_CAP, budget_duration=WINDOW
|
||||
)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
key = client.generate_key(team_id=team_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_drive_to_block(client, key)
|
||||
_poll_until_serves_again(client, key)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.organization.resets_after_window")
|
||||
def test_org_budget_resets_after_window(self, client: BudgetClient, resources: ResourceManager) -> None:
|
||||
org_id = client.create_org(
|
||||
max_budget=TINY_CAP, alias=f"e2e-org-reset-{unique_marker()}", budget_duration=WINDOW
|
||||
)
|
||||
resources.defer(lambda: client.delete_org(org_id))
|
||||
team_id = client.create_team(alias=f"e2e-org-team-{unique_marker()}", organization_id=org_id)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
key = client.generate_key(team_id=team_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
budget_id = client.org_budget_id(org_id)
|
||||
assert budget_id, "org created without a budget row"
|
||||
deadline = time.monotonic() + 30
|
||||
while not any(row.budget_reset_at for row in client.budget_info(budget_id)):
|
||||
if time.monotonic() > deadline:
|
||||
pytest.fail("org budget window never scheduled by the reset job")
|
||||
time.sleep(2)
|
||||
|
||||
_drive_to_block(client, key)
|
||||
_poll_until_serves_again(client, key)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.internal_user.resets_after_window")
|
||||
def test_personal_key_user_budget_resets_after_window(
|
||||
self, client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
user_id = client.create_user(max_budget=TINY_CAP, budget_duration=WINDOW)
|
||||
resources.defer(lambda: client.delete_user(user_id))
|
||||
key = client.generate_key(user_id=user_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_drive_to_block(client, key)
|
||||
_poll_until_serves_again(client, key)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.internal_user.resets_after_window")
|
||||
def test_team_member_key_user_budget_resets_after_window(
|
||||
self, client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
user_id = client.create_user(max_budget=TINY_CAP, budget_duration=WINDOW)
|
||||
resources.defer(lambda: client.delete_user(user_id))
|
||||
team_id = client.create_team(alias=f"e2e-user-team-reset-{unique_marker()}")
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
client.add_team_member(team_id, user_id, max_budget_in_team=100.0)
|
||||
key = client.generate_key(team_id=team_id, user_id=user_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_drive_to_block(client, key)
|
||||
_poll_until_serves_again(client, key)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue