diff --git a/tests/e2e/budgets/BUDGET_CODE_MATRIX.md b/tests/e2e/budgets/BUDGET_CODE_MATRIX.md new file mode 100644 index 00000000000..63cb41228d2 --- /dev/null +++ b/tests/e2e/budgets/BUDGET_CODE_MATRIX.md @@ -0,0 +1,93 @@ +# Budget Code Matrix + +What LiteLLM actually implements for budgets: every entity that can carry a dollar +budget, how the limit is enforced, and where in the code it happens. This is the +"what we support" reference; the companion `BUDGET_TEST_COVERAGE_MATRIX.md` maps +each row to its tests and the e2e gaps. + +Over-budget surfaces as a `budget_exceeded` error (the live suite +`tests/otel_tests/test_e2e_budgeting.py` asserts `type == "budget_exceeded"`, +`code == "429"`); the underlying `BudgetExceededError` is defined in +`litellm/exceptions.py` (`status_code=400`). Enforcement runs in `common_checks()` +/ `auth_checks.py` at auth time, plus pre-call reservation in +`budget_reservation.py`. + +Legend for "Enforced": **block** = request rejected; **filter** = router skips the +deployment; **alert** = notify only, request proceeds. + +--- + +## 1. Per-entity dollar budgets + +| Entity | Budget stored | Hard `max_budget` | Soft budget | Per-window | Model budget | Reset by `budget_duration` | +|--------|---------------|-------------------|-------------|------------|--------------|----------------------------| +| API key | `LiteLLM_VerificationToken` (direct cols + `budget_id` FK) | block (`_virtual_key_max_budget_check`) | alert (`_virtual_key_soft_budget_check`) + 80% alert | block (`_virtual_key_multi_budget_check`) | block (`model_max_budget_limiter.is_key_within_model_budget`) | keys reset job | +| Internal user | `LiteLLM_UserTable` (direct cols) | block (`common_checks`, only when not on a team) | - | - | via `model_max_budget` json | users reset job | +| Team | `LiteLLM_TeamTable` (direct cols) | block (`_team_max_budget_check`) | alert (`_team_soft_budget_check`) | block (`_team_multi_budget_check`) | via `model_max_budget` | teams reset job | +| Team member | `LiteLLM_TeamMembership` -> `LiteLLM_BudgetTable` | block (`_check_team_member_budget`) | - | - | - | budget-table reset job | +| End-user / customer | `LiteLLM_EndUserTable` -> `LiteLLM_BudgetTable` | block (`_check_end_user_budget`) | - | - | block (`is_end_user_within_model_budget`) | budget-table reset job | +| Organization | `LiteLLM_OrganizationTable` -> `LiteLLM_BudgetTable` | block (`_organization_max_budget_check`) | - | - | via budget-table | budget-table reset job | +| Tag | `LiteLLM_TagTable` -> `LiteLLM_BudgetTable` | block (`_tag_max_budget_check`) | - | - | via budget-table | budget-table reset job | +| Project | `LiteLLM_ProjectTable` -> `LiteLLM_BudgetTable` | block (`_project_max_budget_check`) | alert (`_project_soft_budget_check`) | - | - | budget-table reset job | +| Provider (router) | config `provider_budget_config` (in-memory) | filter (`router_strategy/budget_limiter`) | - | yes (time window) | - | window TTL | +| Global proxy | `litellm.max_budget` (config) | block (`_global_proxy_budget_check`) | - | - | - | - | + +Notes / flags from the code: +- **User budget only enforced off-team**: `common_checks` skips the personal-user + budget when the key belongs to a team (team budget governs instead). +- **Comparison operators are inconsistent**: key/user use `>=`, team/end-user main + budget use `>`. Spend exactly at `max_budget` blocks a key but not a team. +- **Provider budgets are filter-only**: an over-budget provider is removed from + routing; if all are over budget the router raises + `no_deployments_with_provider_budget_routing` (not a per-entity block). +- **Enforcement timing differs by entity**: key / user / org / team-member / tag / + model enforce off real-time reservation counters (block within ~2 calls); + **end-user** enforcement reads `EndUserTable.spend`, which only updates on the + `proxy_batch_write_at` flush, so it lags by that interval (verified live). + +## 2. Budget mechanisms + +| Mechanism | What it does | Code | +|-----------|--------------|------| +| Pre-call reservation | Estimates max request cost, atomically reserves against redis spend counters for key/team/user/end_user/tag/team_member/org before the call; blocks if a counter would exceed | `spend_tracking/budget_reservation.py` | +| Post-call reconciliation | Adjusts the reservation to the actual cost once known | `reconcile_budget_reservation` | +| Read-time enforcement | Auth-time check of current spend vs `max_budget` | `auth_checks.common_checks` + per-entity `_*_max_budget_check` | +| Soft budget / alerts | At `soft_budget` (or 80% of max) fire Slack/email alert, do not block | `_virtual_key_soft_budget_check`, `_team_soft_budget_check`, `budget_alerts` | +| Multi-window budgets | `budget_limits` list of `{budget_duration, max_budget}`; each window enforced + reset independently | `_virtual_key_multi_budget_check`, `reset_budget_windows` | +| Model-level budgets | `model_max_budget` dict (per model: `budget_limit` + `time_period`) on key/user/end_user | `hooks/model_max_budget_limiter.py` | +| Reset by duration | Job zeros `spend`, recomputes `budget_reset_at = now + duration_in_seconds(budget_duration)`, invalidates redis counters | `common_utils/reset_budget_job.py`, `duration_parser.duration_in_seconds` | +| Zero-cost bypass | Models with no configured price bypass budget reservation | `budget_reservation` zero-cost path | + +## 3. Budget management surface (endpoints) + +| Action | Endpoint | Handler | +|--------|----------|---------| +| Create budget | `POST /budget/new` | `new_budget` | +| Update budget | `POST /budget/update` | `update_budget` | +| Budget info | `POST /budget/info` (`{"budgets": [id]}`) | `info_budget` | +| Budget settings | `GET /budget/settings` | `budget_settings` | +| List budgets | `GET /budget/list` | `list_budget` | +| Delete budget | `POST /budget/delete` (`{"id": id}`) | `delete_budget` | +| Set on key | `POST /key/generate`, `/key/update` (`max_budget`, `soft_budget`, `budget_duration`, `model_max_budget`, `budget_id`) | key mgmt | +| Set on user | `POST /user/new` (`max_budget`, `budget_duration`) | internal user | +| Set on team | `POST /team/new` (`max_budget`, `soft_budget`, `team_member_budget`) | team | +| Set on team member | `POST /team/member_add` (`max_budget_in_team`) | team | +| Set on org | `POST /organization/new` (`max_budget`, `soft_budget`, `model_max_budget`) | org | +| Set on customer | `POST /customer/new`, `/customer/update` (`max_budget`, `budget_id`) | customer | +| Set on tag | `POST /tag/new`, `/tag/update` (`max_budget`) | tag mgmt | +| Read budget+spend | `/key/info`, `/user/info`, `/team/info`, `/organization/info`, `/customer/info`, `/budget/info` | per-entity info | + +Endpoint method/shape gotchas verified live: `/organization/delete` is **DELETE** +with `{"organization_ids": [id]}`; `/budget/info` takes `{"budgets": [id]}`; +`model_max_budget` entries use `{"budget_limit", "time_period"}`. + +## 4. Config knobs + +| Setting | Effect | +|---------|--------| +| `litellm.max_budget` | proxy-wide hard cap (global proxy budget) | +| `max_internal_user_budget` / `default_max_internal_user_budget` | default `max_budget` for internal users | +| `internal_user_budget_duration` | default reset duration for internal users | +| `max_end_user_budget` / `max_end_user_budget_id` | default budget for end-users | +| `default_team_params` | default `max_budget` / `budget_duration` / limits for teams | +| `provider_budget_config` (router) | per-provider spend caps + windows | diff --git a/tests/e2e/budgets/BUDGET_TEST_COVERAGE_MATRIX.md b/tests/e2e/budgets/BUDGET_TEST_COVERAGE_MATRIX.md new file mode 100644 index 00000000000..62bfc1fdd41 --- /dev/null +++ b/tests/e2e/budgets/BUDGET_TEST_COVERAGE_MATRIX.md @@ -0,0 +1,78 @@ +# Budget Test Coverage Matrix + +Maps every row of `BUDGET_CODE_MATRIX.md` (what LiteLLM implements) to its tests +and level, then marks the live e2e coverage this suite adds. + +Levels: `unit` mocked (`AsyncMock` on `get_current_spend`/prisma); `router` live +router with fake deployments; `live-e2e` real proxy, real key/team, real requests +until blocked. Status: `covered` / `partial` / `gap`. + +Pre-existing live coverage outside this suite: +- `tests/otel_tests/test_e2e_budgeting.py` - key + team enforcement, budget update. +- `tests/local_testing/test_router_budget_limiter.py` - provider / tag / deployment + budgets at the router. + +This suite (`tests/e2e/budgets/`) adds the missing live coverage and runs +on the shared lifecycle (every entity it creates is deleted on teardown). + +--- + +## Per-entity enforcement + +| Entity | Unit | Pre-existing live | This suite (live) | Status | +|--------|------|-------------------|-------------------|--------| +| API key | `test_budget_reservation.py`, `test_max_budget_limiter.py` | `otel_tests` | `test_budget_enforcement_e2e::test_key_budget_blocks` | **covered** | +| Team | `test_team_budget_limits.py` | `otel_tests` | (org test builds a team) | **covered** | +| Internal user | auth unit tests | - | `test_internal_user_budget_blocks` | **covered (new)** | +| Team member | `test_team_member_budget.py` | - | `test_team_member_budget_blocks` | **covered (new)** | +| End-user / customer | `test_custom_auth_end_user_budget.py` | - | `test_end_user_budget_blocks` | **covered (new)** | +| Organization | `test_organization_budget_enforcement.py` (flagged weak) | - | `test_organization_budget_blocks` | **covered (new)** | +| Tag (proxy-level) | - | router only | `test_tag_budget_e2e::test_tag_budget_blocks_tagged_requests` | **covered (new)** | +| Model-level (`model_max_budget`) | `test_unit_test_max_model_budget_limiter.py` | - | `test_model_max_budget_e2e::test_model_max_budget_isolates_per_model` | **covered (new)** | +| Provider (router) | `test_budget_limiter_hotpath.py` | `test_router_budget_limiter.py` | - | **covered** (router) | +| Global proxy (`litellm.max_budget`) | unit | - | - | **gap** (needs a config-level cap; not key-settable) | + +## Budget mechanisms + +| Mechanism | Unit | This suite (live) | Status | +|-----------|------|-------------------|--------| +| Pre-call reservation | `test_budget_reservation.py` | exercised by every enforcement test | **partial** | +| Soft budget / alerts | `SlackAlerting/test_budget_alert_types.py` | `test_soft_budget_e2e::test_soft_budget_does_not_block` | **covered (new)** (block-vs-alert; the alert side-effect itself stays unit) | +| Budget CRUD | `test_budget_endpoints.py` | `test_budget_crud_e2e` (roundtrip + delete) | **covered (new)** | +| Reset scheduling | `test_proxy_budget_reset.py` | `test_budget_crud_e2e::test_budget_duration_schedules_reset_on_key` | **covered (new)** (scheduling; actual zeroing is time-dependent -> unit) | +| Multi-window budgets | `test_multi_budget_windows.py` | - | **gap** (window setup is fiddly; left to unit for now) | +| Read budget+spend | `test_spend_management_endpoints.py` | `/key/info` asserted in CRUD + enforcement | **partial** | + +## Remaining gaps (intentionally not live-tested) + +- **Global proxy budget** (`litellm.max_budget`): set via proxy config, not a + per-key API, so it needs a dedicated proxy boot with that config rather than a + runtime-created entity. Out of scope for the per-entity suite. +- **Multi-window budgets**: the `budget_limits` list shape and per-window reset are + covered by `test_multi_budget_windows.py` (unit); a live version would need to + wait out a short window to see the reset, which is time-dependent. +- **Soft-budget alert delivery**: whether the Slack/email actually fires is not + observable from the proxy API; unit tests own that. The live test pins the + load-bearing behavior (soft does not block). +- **Reset zeroing after the window elapses**: time-dependent; unit tests own the + reset-job logic. The live test pins that `budget_reset_at` is scheduled. + +## This suite's files + +| File | Covers | +|------|--------| +| `test_budget_enforcement_e2e.py` | key / internal-user / end-user / organization / team-member hard enforcement | +| `test_model_max_budget_e2e.py` | per-model caps isolate by model | +| `test_soft_budget_e2e.py` | soft budget alerts but does not block | +| `test_tag_budget_e2e.py` | proxy-level tag budget blocks tagged requests, spares others | +| `test_budget_crud_e2e.py` | `/budget/*` CRUD roundtrip + delete + `budget_reset_at` scheduling | + +## Pattern + timing + +Create the entity with a tiny `max_budget`, drive spend until a `budget_exceeded` +block. The enforcement helper is two-phase: a fast warmup (key/user/org/member/tag/ +model block within ~2 calls off real-time counters), then a poll across the ~60s +batch-write window (end-user enforcement reads table spend that lags). Skip on a +non-budget error (provider down / key missing); fail if the budget is never +enforced. Chat tests use `gpt-5.5` (the model with a working key on the reference +proxy); swap the literal if your proxy differs. diff --git a/tests/e2e/budgets/budget_client.py b/tests/e2e/budgets/budget_client.py new file mode 100644 index 00000000000..1f2c6039973 --- /dev/null +++ b/tests/e2e/budgets/budget_client.py @@ -0,0 +1,166 @@ +"""Client for budget e2e tests: the shared ProxyClient plus budget-bearing entity +management (user / team / team-member / org / customer / tag / budget-table) and +info reads. + +Over-budget surfaces as a ``budget_exceeded`` error; ``is_budget_block`` detects it +on a CallResult. Create methods return the new id and raise on failure; tests +register the matching delete with ``resources.defer(...)`` for cleanup. +""" + +from typing import Dict, Optional + +import requests +from pydantic import TypeAdapter +from pydantic.dataclasses import dataclass + +from proxy_client import CallResult, ProxyClient, auth_headers, proxy_client_kwargs + + +def is_budget_block(result: CallResult) -> bool: + """True if the call was rejected for being over budget (vs a provider error).""" + return not result.ok and "budget_exceeded" in result.body + + +def model_budget(model: str, limit: float, period: str = "30d") -> dict: + """A model_max_budget dict entry: per-model cap with a reset window.""" + return {model: {"budget_limit": limit, "time_period": period}} + + +@dataclass(frozen=True, slots=True) +class BudgetRow: + """A /budget/info row: only the fields tests assert on, pydantic ignores the rest.""" + + budget_id: Optional[str] = None + max_budget: Optional[float] = None + soft_budget: Optional[float] = None + budget_duration: Optional[str] = None + budget_reset_at: Optional[str] = None + + +_BUDGET_ROWS = TypeAdapter(tuple[BudgetRow, ...]) + + +class BudgetClient(ProxyClient): + def _post(self, path: str, body: Dict[str, object]) -> Dict[str, object]: + resp = requests.post( + f"{self._base_url}{path}", + headers=auth_headers(self._master_key), + json=body, + timeout=self._request_timeout, + ) + resp.raise_for_status() + data = resp.json() if resp.text else {} + # delete routes return a bare count, not an object; callers ignore it. + return data if isinstance(data, dict) else {} + + def _delete(self, path: str, body: Dict[str, object]) -> None: + resp = requests.delete( + f"{self._base_url}{path}", + headers=auth_headers(self._master_key), + json=body, + timeout=self._request_timeout, + ) + resp.raise_for_status() + + # ---- internal user -------------------------------------------------- + + def create_user(self, *, max_budget: float, budget_duration: Optional[str] = None) -> str: + body: Dict[str, object] = {"max_budget": max_budget} + if budget_duration is not None: + body["budget_duration"] = budget_duration + return str(self._post("/user/new", body)["user_id"]) + + def delete_user(self, user_id: str) -> None: + self._post("/user/delete", {"user_ids": [user_id]}) + + # ---- customer / end-user ------------------------------------------- + + def create_customer(self, customer_id: str, *, max_budget: float) -> str: + self._post("/customer/new", {"user_id": customer_id, "max_budget": max_budget}) + return customer_id + + # ---- organization --------------------------------------------------- + + def create_org(self, *, max_budget: float, alias: str) -> str: + return str( + self._post( + "/organization/new", + {"organization_alias": alias, "max_budget": max_budget}, + )["organization_id"] + ) + + def delete_org(self, org_id: str) -> None: + self._delete("/organization/delete", {"organization_ids": [org_id]}) + + # ---- team ----------------------------------------------------------- + + def create_team( + self, + *, + alias: str, + max_budget: Optional[float] = None, + organization_id: Optional[str] = None, + extra: Optional[Dict[str, object]] = None, + ) -> str: + body: Dict[str, object] = {"team_alias": alias} + if max_budget is not None: + body["max_budget"] = max_budget + if organization_id is not None: + body["organization_id"] = organization_id + if extra: + body.update(extra) + return str(self._post("/team/new", body)["team_id"]) + + def delete_team(self, team_id: str) -> None: + self._post("/team/delete", {"team_ids": [team_id]}) + + def add_team_member(self, team_id: str, user_id: str, *, max_budget_in_team: Optional[float] = None) -> None: + body: Dict[str, object] = { + "team_id": team_id, + "member": {"role": "user", "user_id": user_id}, + } + if max_budget_in_team is not None: + body["max_budget_in_team"] = max_budget_in_team + self._post("/team/member_add", body) + + # ---- tag ------------------------------------------------------------ + + def create_tag(self, name: str, *, max_budget: float) -> str: + self._post("/tag/new", {"name": name, "max_budget": max_budget}) + return name + + def delete_tag(self, name: str) -> None: + self._post("/tag/delete", {"name": name}) + + # ---- budget table --------------------------------------------------- + + def create_budget( + self, + *, + max_budget: float, + soft_budget: Optional[float] = None, + budget_duration: Optional[str] = None, + ) -> str: + body: Dict[str, object] = {"max_budget": max_budget} + if soft_budget is not None: + body["soft_budget"] = soft_budget + if budget_duration is not None: + body["budget_duration"] = budget_duration + return str(self._post("/budget/new", body)["budget_id"]) + + def delete_budget(self, budget_id: str) -> None: + self._post("/budget/delete", {"id": budget_id}) + + def budget_info(self, budget_id: str) -> tuple[BudgetRow, ...]: + resp = requests.post( + f"{self._base_url}/budget/info", + headers=auth_headers(self._master_key), + json={"budgets": [budget_id]}, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return _BUDGET_ROWS.validate_python(resp.json()) + + +def build_client() -> BudgetClient: + return BudgetClient(**proxy_client_kwargs()) diff --git a/tests/e2e/budgets/conftest.py b/tests/e2e/budgets/conftest.py new file mode 100644 index 00000000000..88c4cae478f --- /dev/null +++ b/tests/e2e/budgets/conftest.py @@ -0,0 +1,16 @@ +"""Budgets suite's `client` fixture. + +The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker +live in the parent tests/e2e/conftest.py. BudgetClient subclasses +ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources` +fixture cleans up keys; tests register entity deletes via `resources.defer(...)`. +""" + +import pytest + +from budget_client import BudgetClient, build_client + + +@pytest.fixture(scope="session") +def client() -> BudgetClient: + return build_client() diff --git a/tests/e2e/budgets/test_budget_crud_e2e.py b/tests/e2e/budgets/test_budget_crud_e2e.py new file mode 100644 index 00000000000..3a2d1dd4fc8 --- /dev/null +++ b/tests/e2e/budgets/test_budget_crud_e2e.py @@ -0,0 +1,59 @@ +"""Live e2e for the budget management surface (no LLM calls, fast). + +Covers the budget-table CRUD round-trip and that `budget_duration` schedules a +`budget_reset_at`. The actual zeroing after the window is time-dependent, so we +assert the reset is *scheduled* (now + duration), not waited out. +""" + +from datetime import datetime, timezone + +import pytest + +from budget_client import BudgetClient +from lifecycle import ResourceManager + +pytestmark = pytest.mark.e2e + + +def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None: + budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d") + resources.defer(lambda: client.delete_budget(budget_id)) + + rows = client.budget_info(budget_id) + assert rows, f"/budget/info returned nothing for {budget_id}" + row = rows[0] + assert row.max_budget == 12.5 + assert row.soft_budget == 10.0 + assert row.budget_reset_at, "budget_duration did not schedule a reset" + + # Attach the budget to a key and confirm the key reflects it. + key = client.generate_key(extra_params={"budget_id": budget_id}) + resources.defer(lambda: client.delete_key(key)) + info = client.key_info(key) + linked = info.get("litellm_budget_table") or {} + assert info.get("budget_id") == budget_id or linked.get("max_budget") == 12.5, ( + f"key does not reflect attached budget: {info.get('budget_id')}, {linked}" + ) + + +def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None: + budget_id = client.create_budget(max_budget=1.0) + client.delete_budget(budget_id) + assert not client.budget_info(budget_id), "budget still present after delete" + + +def test_budget_duration_schedules_reset_on_key(client: BudgetClient, resources: ResourceManager) -> None: + key = client.generate_key(max_budget=10.0, extra_params={"budget_duration": "30d"}) + resources.defer(lambda: client.delete_key(key)) + + reset_at = client.key_info(key).get("budget_reset_at") + assert reset_at, "budget_duration did not set budget_reset_at on the key" + + # budget_duration schedules a FUTURE reset. Don't assume now+30d exactly: the + # proxy may align the reset to a calendar boundary (e.g. start of next month), + # so "30d" can land ~12 days out mid-month. Assert it's scheduled ahead. + + # get current time -> assert budget from days_left - budget_duration == days_left + reset_dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00")) + days_out = (reset_dt - datetime.now(timezone.utc)).total_seconds() / 86400 + assert 0 < days_out < 40, f"reset should be scheduled ahead, got {days_out:.1f}d out" diff --git a/tests/e2e/budgets/test_budget_enforcement_e2e.py b/tests/e2e/budgets/test_budget_enforcement_e2e.py new file mode 100644 index 00000000000..c68cdf872ee --- /dev/null +++ b/tests/e2e/budgets/test_budget_enforcement_e2e.py @@ -0,0 +1,142 @@ +"""Live e2e: a tiny max_budget on an entity actually blocks requests. + +Each entity is an E2ECase (lifecycle.E2ECase) driven by run_case: init() creates +the budgeted entity + a key, run() drives spend until a `budget_exceeded` block, +teardown() deletes everything init() created (always runs, even on failure/skip). +Covers the entities with no prior live coverage - internal user, end-user, +organization, team member. See BUDGET_TEST_COVERAGE_MATRIX.md. + +A non-budget error fails hard (never a skip); if calls never get blocked, budget +enforcement is broken -> fail. +""" + +import time +from dataclasses import dataclass, field +from typing import Callable, Dict, List, Type + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import run_case +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") -> None: + """Send paid calls until the entity's budget blocks one. Key/user/org/member + block within a couple calls off real-time reservation counters; the end-user + budget enforces off table spend that lands on the batch write, so it takes a + few more. A non-budget error fails hard (never a skip).""" + extra: Dict[str, object] = {"max_tokens": 16} + if user: + extra["user"] = user + for _ in range(40): + result = client.chat(key, "claude-haiku-4-5", f"spend {unique_marker()}", extra_body=extra) + if is_budget_block(result): + return + require_successful_call(result) + time.sleep(2) + pytest.fail("budget never enforced within the call budget") + + +@dataclass +class _BudgetCase: + """Base E2ECase: a key under some budgeted entity must get blocked. + + Subclasses set up the budgeted entity in init() and register every created id + in `_undo` (run LIFO in teardown so a key is deleted before its team/org). + """ + + client: BudgetClient + key: str = "" + _undo: List[Callable[[], None]] = field( + default_factory=list + ) # mutable-ok: per-case teardown registry + + def init(self) -> None: + raise NotImplementedError + + def run(self) -> None: + _assert_budget_blocks(self.client, self.key) + + def teardown(self) -> None: + for undo in reversed(self._undo): + undo() + + +class KeyBudgetCase(_BudgetCase): + def init(self) -> None: + self.key = self.client.generate_key(max_budget=3e-6) + self._undo.append(lambda: self.client.delete_key(self.key)) + + +class InternalUserBudgetCase(_BudgetCase): + def init(self) -> None: + user_id = self.client.create_user(max_budget=3e-6) + self._undo.append(lambda: self.client.delete_user(user_id)) + # personal key (no team) -> the user budget governs + self.key = self.client.generate_key(extra_params={"user_id": user_id}) + self._undo.append(lambda: self.client.delete_key(self.key)) + + +class EndUserBudgetCase(_BudgetCase): + def init(self) -> None: + customer = f"e2e-budget-cust-{unique_marker()}" + self.client.create_customer(customer, max_budget=3e-6) + self._undo.append(lambda: self.client.delete_customers([customer])) + self.key = self.client.generate_key(models=["claude-haiku-4-5"]) + self._undo.append(lambda: self.client.delete_key(self.key)) + self._customer = customer + + def run(self) -> None: + _assert_budget_blocks(self.client, self.key, user=self._customer) + + +class OrganizationBudgetCase(_BudgetCase): + def init(self) -> None: + # Org carries the tiny budget; the team under it has none, so a block here + # is org-level enforcement (the historically weak link). + org_id = self.client.create_org( + max_budget=3e-6, alias=f"e2e-budget-org-{unique_marker()}" + ) + self._undo.append(lambda: self.client.delete_org(org_id)) + team_id = self.client.create_team( + alias=f"e2e-budget-team-{unique_marker()}", organization_id=org_id + ) + self._undo.append(lambda: self.client.delete_team(team_id)) + self.key = self.client.generate_key(extra_params={"team_id": team_id}) + self._undo.append(lambda: self.client.delete_key(self.key)) + + +class TeamMemberBudgetCase(_BudgetCase): + def init(self) -> None: + # Member's per-team budget is tiny while the team has a large budget, so a + # block proves member-level (not team-level) enforcement. + team_id = self.client.create_team( + alias=f"e2e-budget-team-{unique_marker()}", max_budget=100.0 + ) + self._undo.append(lambda: self.client.delete_team(team_id)) + user_id = self.client.create_user(max_budget=100.0) + self._undo.append(lambda: self.client.delete_user(user_id)) + self.client.add_team_member(team_id, user_id, max_budget_in_team=3e-6) + self.key = self.client.generate_key( + extra_params={"team_id": team_id, "user_id": user_id} + ) + self._undo.append(lambda: self.client.delete_key(self.key)) + + +@pytest.mark.parametrize( + "case_cls", + [ + KeyBudgetCase, + InternalUserBudgetCase, + EndUserBudgetCase, + OrganizationBudgetCase, + TeamMemberBudgetCase, + ], + ids=lambda c: c.__name__, +) +def test_budget_enforcement( + client: BudgetClient, case_cls: Type[_BudgetCase] +) -> None: + run_case(case_cls(client)) diff --git a/tests/e2e/budgets/test_budget_reset_e2e.py b/tests/e2e/budgets/test_budget_reset_e2e.py new file mode 100644 index 00000000000..35923e71410 --- /dev/null +++ b/tests/e2e/budgets/test_budget_reset_e2e.py @@ -0,0 +1,55 @@ +"""Live e2e: a key budget resets (zeroes spend) after its budget_duration. + +Short budget_duration (30s) + the fast-rescheduled reset job: a key blocked for +exceeding its max_budget starts succeeding again once the duration elapses and the +reset job zeroes key.spend. Closes the reset-zeroing gap in +BUDGET_TEST_COVERAGE_MATRIX.md (reset_budget_for_litellm_keys), which the unit +suite covers but no live test did - distinct from the per-window reset in +test_multi_window_budget_e2e.py. +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + + +def _call(client: BudgetClient, key: str): + return client.chat( + key, "claude-haiku-4-5", f"reset {unique_marker()}", extra_body={"max_tokens": 16} + ) + + +def test_key_budget_resets_after_duration( + client: BudgetClient, resources: ResourceManager +) -> None: + key = client.generate_key(max_budget=3e-6, extra_params={"budget_duration": "30s"}) + resources.defer(lambda: client.delete_key(key)) + + # 1. exceed the budget -> litellm returns budget_exceeded + blocked = False + for _ in range(20): + result = _call(client, key) + if is_budget_block(result): + blocked = True + break + require_successful_call(result) + time.sleep(2) + assert blocked, "key budget never enforced" + + # 2. once the 30s duration elapses + the reset job runs, key.spend zeroes and + # calls flow again, within a span only a short duration could produce. + start = time.monotonic() + while time.monotonic() < start + 90: + time.sleep(5) + result = _call(client, key) + if result.ok: + assert time.monotonic() - start < 75, "reset too slow for a 30s budget" + return + assert is_budget_block(result), f"non-budget error: {result.body[:200]}" + pytest.fail("key budget never reset within 90s") diff --git a/tests/e2e/budgets/test_model_max_budget_e2e.py b/tests/e2e/budgets/test_model_max_budget_e2e.py new file mode 100644 index 00000000000..7ed3e0c2834 --- /dev/null +++ b/tests/e2e/budgets/test_model_max_budget_e2e.py @@ -0,0 +1,60 @@ +"""Live e2e: per-model budgets (`model_max_budget`) isolate by model. + +A key caps one model tiny and leaves another generous. Exhausting the capped +model must block *that* model while the other still works - proving the per-model +cap is enforced independently, not as a key-wide budget. Closes the +model_max_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block, model_budget +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +CAPPED_MODEL = "claude-haiku-4-5" +FREE_MODEL = "gemini-2.5-flash" + + +def _call(client: BudgetClient, key: str, model: str): + result = client.chat( + key, model, f"hi {unique_marker()}", extra_body={"max_tokens": 16} + ) + if not result.ok and not is_budget_block(result): + require_successful_call(result) # non-budget error -> skip + return result + + +def test_model_max_budget_isolates_per_model( + client: BudgetClient, resources: ResourceManager +) -> None: + key = client.generate_key( + extra_params={ + "model_max_budget": { + **model_budget(CAPPED_MODEL, 1e-6), + **model_budget(FREE_MODEL, 1000.0), + } + } + ) + resources.defer(lambda: client.delete_key(key)) + + # Exhaust the capped model. + blocked = False + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + if is_budget_block(_call(client, key, CAPPED_MODEL)): + blocked = True + break + time.sleep(1) + assert blocked, f"{CAPPED_MODEL} per-model budget never enforced" + + # The other model shares the key but has its own (large) cap -> still works. + other = _call(client, key, FREE_MODEL) + require_successful_call(other) # its own large cap -> must succeed, never a budget block + assert not is_budget_block(other), ( + f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" + ) diff --git a/tests/e2e/budgets/test_multi_window_budget_e2e.py b/tests/e2e/budgets/test_multi_window_budget_e2e.py new file mode 100644 index 00000000000..b9200a1ea32 --- /dev/null +++ b/tests/e2e/budgets/test_multi_window_budget_e2e.py @@ -0,0 +1,70 @@ +"""Live e2e: multi-window budgets (budget_limits) enforce AND reset per window. + +Short windows make the time limit reachable inside a test: a tight 30s window and +a roomy 1m window. The 30s window blocks once its tiny cap is exceeded, then - once +its 30s elapses and the reset job runs (rescheduled fast via +PROXY_BUDGET_RESCHEDULER_* in docker-compose) - the window resets and calls flow +again. Closes the multi-window gap (enforcement + per-window reset) in +BUDGET_TEST_COVERAGE_MATRIX.md, which the unit suite covered but no live test did. +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +WINDOW_SECONDS = 30 # the tight window; calls succeed again only after it elapses + + +def _call(client: BudgetClient, key: str): + return client.chat( + key, "claude-haiku-4-5", f"window {unique_marker()}", extra_body={"max_tokens": 16} + ) + + +def test_short_window_blocks_then_resets( + client: BudgetClient, resources: ResourceManager +) -> None: + key = client.generate_key( + extra_params={ + "budget_limits": [ + {"budget_duration": f"{WINDOW_SECONDS}s", "max_budget": 3e-6}, + {"budget_duration": "1m", "max_budget": 1.0}, # roomy: never blocks + ] + } + ) + resources.defer(lambda: client.delete_key(key)) + + # 1. exhaust the tight window -> litellm returns budget_exceeded + start = time.monotonic() + blocked = False + for _ in range(20): + result = _call(client, key) + if is_budget_block(result): + blocked = True + break + require_successful_call(result) + time.sleep(2) + assert blocked, f"{WINDOW_SECONDS}s window never enforced" + + # 2. the window resets at the next wall-clock-aligned boundary (so it can land + # a little under WINDOW_SECONDS from creation) + the reset job. When a call + # flows again the window has reset; the elapsed clock must be short enough + # that this is the 30s window resetting, not the roomy 1m one. + deadline = time.monotonic() + 90 + while time.monotonic() < deadline: + time.sleep(5) + result = _call(client, key) + if result.ok: + elapsed = time.monotonic() - start + assert elapsed < WINDOW_SECONDS + 45, ( + f"reset took {elapsed:.0f}s - too long for a {WINDOW_SECONDS}s window" + ) + return + assert is_budget_block(result), f"non-budget error during reset wait: {result.body[:200]}" + pytest.fail(f"{WINDOW_SECONDS}s window never reset within 90s") diff --git a/tests/e2e/budgets/test_soft_budget_e2e.py b/tests/e2e/budgets/test_soft_budget_e2e.py new file mode 100644 index 00000000000..51b1a99a1dc --- /dev/null +++ b/tests/e2e/budgets/test_soft_budget_e2e.py @@ -0,0 +1,36 @@ +"""Live e2e: soft_budget alerts but does NOT block. + +A key with a tiny `soft_budget` well under a large `max_budget`: spend crosses the +soft threshold within a couple calls, but requests keep succeeding (soft budget is +advisory). Closes the soft_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. The alert +side-effect (Slack/email) is not observable from the proxy API, so we assert the +load-bearing behavior: soft != block. +""" + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + + +def test_soft_budget_does_not_block( + client: BudgetClient, resources: ResourceManager +) -> None: + # soft far below max: spend crosses soft immediately, stays under max. + key = client.generate_key( + max_budget=1000.0, extra_params={"soft_budget": 1e-9} + ) + resources.defer(lambda: client.delete_key(key)) + + for _ in range(3): + result = client.chat( + key, "claude-haiku-4-5", f"hi {unique_marker()}", extra_body={"max_tokens": 16} + ) + require_successful_call(result) # skip if provider unavailable + assert not is_budget_block(result), ( + "soft_budget blocked a request; it must alert only, not block " + f"(body={result.body[:200]})" + ) diff --git a/tests/e2e/budgets/test_tag_budget_e2e.py b/tests/e2e/budgets/test_tag_budget_e2e.py new file mode 100644 index 00000000000..77a3ff35eca --- /dev/null +++ b/tests/e2e/budgets/test_tag_budget_e2e.py @@ -0,0 +1,58 @@ +"""Live e2e: proxy-level tag budgets block tagged requests. + +A tag with a tiny budget: requests carrying that tag get blocked once the tag's +spend is exceeded, while a request with a different tag (no budget) still works. +Closes the proxy-level tag-budget gap in BUDGET_TEST_COVERAGE_MATRIX.md (today +only router-level tag budgets are tested). +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +TINY_BUDGET = 1e-6 + + +def _tagged_call(client: BudgetClient, key: str, tag: str): + result = client.chat( + key, + "claude-haiku-4-5", + f"hi {unique_marker()}", + metadata={"tags": [tag]}, + extra_body={"max_tokens": 16}, + ) + if not result.ok and not is_budget_block(result): + require_successful_call(result) # non-budget error -> skip + return result + + +def test_tag_budget_blocks_tagged_requests( + client: BudgetClient, scoped_key: str, resources: ResourceManager +) -> None: + budgeted_tag = f"e2e-budget-tag-{unique_marker()}" + client.create_tag(budgeted_tag, max_budget=TINY_BUDGET) + resources.defer(lambda: client.delete_tag(budgeted_tag)) + + # Requests under the budgeted tag get blocked once its spend is exceeded. + blocked = False + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + if is_budget_block(_tagged_call(client, scoped_key, budgeted_tag)): + blocked = True + break + time.sleep(1) + assert blocked, f"tag budget for {budgeted_tag!r} never enforced" + + # A request with an unbudgeted tag on the same key is unaffected. + free_tag = f"e2e-free-tag-{unique_marker()}" + other = _tagged_call(client, scoped_key, free_tag) + require_successful_call(other) + assert not is_budget_block(other), ( + f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget" + )