From 8786e301bfcb8984fc0a6d79120ae1be6eb2992c Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Thu, 18 Jun 2026 19:41:11 -0700 Subject: [PATCH] remove e2e_tests folder --- tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md | 93 ----- .../budgets/BUDGET_TEST_COVERAGE_MATRIX.md | 78 ---- tests/e2e_tests/budgets/budget_client.py | 185 --------- tests/e2e_tests/budgets/conftest.py | 16 - .../e2e_tests/budgets/test_budget_crud_e2e.py | 68 ---- .../budgets/test_budget_enforcement_e2e.py | 119 ------ .../budgets/test_model_max_budget_e2e.py | 60 --- .../e2e_tests/budgets/test_soft_budget_e2e.py | 36 -- .../e2e_tests/budgets/test_tag_budget_e2e.py | 58 --- tests/e2e_tests/conftest.py | 54 --- tests/e2e_tests/docker-compose.yml | 59 --- tests/e2e_tests/docker-config.yaml | 39 -- tests/e2e_tests/e2e_config.py | 16 - tests/e2e_tests/lifecycle.py | 87 ---- .../LLM_TRANSLATION_COVERAGE_MATRIX.md | 84 ---- tests/e2e_tests/llm_translation/conftest.py | 16 - .../llm_translation/passthrough_client.py | 136 ------- .../llm_translation/test_passthrough_e2e.py | 158 -------- tests/e2e_tests/proxy_client.py | 371 ------------------ tests/e2e_tests/pytest.ini | 7 - .../SPEND_TRACKING_COVERAGE_MATRIX.md | 77 ---- tests/e2e_tests/spend_tracking/conftest.py | 16 - .../spend_tracking/spend_e2e_client.py | 94 ----- .../spend_tracking/test_spend_routes.py | 93 ----- .../spend_tracking/test_spend_tracking_e2e.py | 335 ---------------- 25 files changed, 2355 deletions(-) delete mode 100644 tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md delete mode 100644 tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md delete mode 100644 tests/e2e_tests/budgets/budget_client.py delete mode 100644 tests/e2e_tests/budgets/conftest.py delete mode 100644 tests/e2e_tests/budgets/test_budget_crud_e2e.py delete mode 100644 tests/e2e_tests/budgets/test_budget_enforcement_e2e.py delete mode 100644 tests/e2e_tests/budgets/test_model_max_budget_e2e.py delete mode 100644 tests/e2e_tests/budgets/test_soft_budget_e2e.py delete mode 100644 tests/e2e_tests/budgets/test_tag_budget_e2e.py delete mode 100644 tests/e2e_tests/conftest.py delete mode 100644 tests/e2e_tests/docker-compose.yml delete mode 100644 tests/e2e_tests/docker-config.yaml delete mode 100644 tests/e2e_tests/e2e_config.py delete mode 100644 tests/e2e_tests/lifecycle.py delete mode 100644 tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md delete mode 100644 tests/e2e_tests/llm_translation/conftest.py delete mode 100644 tests/e2e_tests/llm_translation/passthrough_client.py delete mode 100644 tests/e2e_tests/llm_translation/test_passthrough_e2e.py delete mode 100644 tests/e2e_tests/proxy_client.py delete mode 100644 tests/e2e_tests/pytest.ini delete mode 100644 tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md delete mode 100644 tests/e2e_tests/spend_tracking/conftest.py delete mode 100644 tests/e2e_tests/spend_tracking/spend_e2e_client.py delete mode 100644 tests/e2e_tests/spend_tracking/test_spend_routes.py delete mode 100644 tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py diff --git a/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md b/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md deleted file mode 100644 index 63cb41228d2..00000000000 --- a/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md +++ /dev/null @@ -1,93 +0,0 @@ -# Budget Code Matrix - -What LiteLLM actually implements for budgets: every entity that can carry a dollar -budget, how the limit is enforced, and where in the code it happens. This is the -"what we support" reference; the companion `BUDGET_TEST_COVERAGE_MATRIX.md` maps -each row to its tests and the e2e gaps. - -Over-budget surfaces as a `budget_exceeded` error (the live suite -`tests/otel_tests/test_e2e_budgeting.py` asserts `type == "budget_exceeded"`, -`code == "429"`); the underlying `BudgetExceededError` is defined in -`litellm/exceptions.py` (`status_code=400`). Enforcement runs in `common_checks()` -/ `auth_checks.py` at auth time, plus pre-call reservation in -`budget_reservation.py`. - -Legend for "Enforced": **block** = request rejected; **filter** = router skips the -deployment; **alert** = notify only, request proceeds. - ---- - -## 1. Per-entity dollar budgets - -| Entity | Budget stored | Hard `max_budget` | Soft budget | Per-window | Model budget | Reset by `budget_duration` | -|--------|---------------|-------------------|-------------|------------|--------------|----------------------------| -| API key | `LiteLLM_VerificationToken` (direct cols + `budget_id` FK) | block (`_virtual_key_max_budget_check`) | alert (`_virtual_key_soft_budget_check`) + 80% alert | block (`_virtual_key_multi_budget_check`) | block (`model_max_budget_limiter.is_key_within_model_budget`) | keys reset job | -| Internal user | `LiteLLM_UserTable` (direct cols) | block (`common_checks`, only when not on a team) | - | - | via `model_max_budget` json | users reset job | -| Team | `LiteLLM_TeamTable` (direct cols) | block (`_team_max_budget_check`) | alert (`_team_soft_budget_check`) | block (`_team_multi_budget_check`) | via `model_max_budget` | teams reset job | -| Team member | `LiteLLM_TeamMembership` -> `LiteLLM_BudgetTable` | block (`_check_team_member_budget`) | - | - | - | budget-table reset job | -| End-user / customer | `LiteLLM_EndUserTable` -> `LiteLLM_BudgetTable` | block (`_check_end_user_budget`) | - | - | block (`is_end_user_within_model_budget`) | budget-table reset job | -| Organization | `LiteLLM_OrganizationTable` -> `LiteLLM_BudgetTable` | block (`_organization_max_budget_check`) | - | - | via budget-table | budget-table reset job | -| Tag | `LiteLLM_TagTable` -> `LiteLLM_BudgetTable` | block (`_tag_max_budget_check`) | - | - | via budget-table | budget-table reset job | -| Project | `LiteLLM_ProjectTable` -> `LiteLLM_BudgetTable` | block (`_project_max_budget_check`) | alert (`_project_soft_budget_check`) | - | - | budget-table reset job | -| Provider (router) | config `provider_budget_config` (in-memory) | filter (`router_strategy/budget_limiter`) | - | yes (time window) | - | window TTL | -| Global proxy | `litellm.max_budget` (config) | block (`_global_proxy_budget_check`) | - | - | - | - | - -Notes / flags from the code: -- **User budget only enforced off-team**: `common_checks` skips the personal-user - budget when the key belongs to a team (team budget governs instead). -- **Comparison operators are inconsistent**: key/user use `>=`, team/end-user main - budget use `>`. Spend exactly at `max_budget` blocks a key but not a team. -- **Provider budgets are filter-only**: an over-budget provider is removed from - routing; if all are over budget the router raises - `no_deployments_with_provider_budget_routing` (not a per-entity block). -- **Enforcement timing differs by entity**: key / user / org / team-member / tag / - model enforce off real-time reservation counters (block within ~2 calls); - **end-user** enforcement reads `EndUserTable.spend`, which only updates on the - `proxy_batch_write_at` flush, so it lags by that interval (verified live). - -## 2. Budget mechanisms - -| Mechanism | What it does | Code | -|-----------|--------------|------| -| Pre-call reservation | Estimates max request cost, atomically reserves against redis spend counters for key/team/user/end_user/tag/team_member/org before the call; blocks if a counter would exceed | `spend_tracking/budget_reservation.py` | -| Post-call reconciliation | Adjusts the reservation to the actual cost once known | `reconcile_budget_reservation` | -| Read-time enforcement | Auth-time check of current spend vs `max_budget` | `auth_checks.common_checks` + per-entity `_*_max_budget_check` | -| Soft budget / alerts | At `soft_budget` (or 80% of max) fire Slack/email alert, do not block | `_virtual_key_soft_budget_check`, `_team_soft_budget_check`, `budget_alerts` | -| Multi-window budgets | `budget_limits` list of `{budget_duration, max_budget}`; each window enforced + reset independently | `_virtual_key_multi_budget_check`, `reset_budget_windows` | -| Model-level budgets | `model_max_budget` dict (per model: `budget_limit` + `time_period`) on key/user/end_user | `hooks/model_max_budget_limiter.py` | -| Reset by duration | Job zeros `spend`, recomputes `budget_reset_at = now + duration_in_seconds(budget_duration)`, invalidates redis counters | `common_utils/reset_budget_job.py`, `duration_parser.duration_in_seconds` | -| Zero-cost bypass | Models with no configured price bypass budget reservation | `budget_reservation` zero-cost path | - -## 3. Budget management surface (endpoints) - -| Action | Endpoint | Handler | -|--------|----------|---------| -| Create budget | `POST /budget/new` | `new_budget` | -| Update budget | `POST /budget/update` | `update_budget` | -| Budget info | `POST /budget/info` (`{"budgets": [id]}`) | `info_budget` | -| Budget settings | `GET /budget/settings` | `budget_settings` | -| List budgets | `GET /budget/list` | `list_budget` | -| Delete budget | `POST /budget/delete` (`{"id": id}`) | `delete_budget` | -| Set on key | `POST /key/generate`, `/key/update` (`max_budget`, `soft_budget`, `budget_duration`, `model_max_budget`, `budget_id`) | key mgmt | -| Set on user | `POST /user/new` (`max_budget`, `budget_duration`) | internal user | -| Set on team | `POST /team/new` (`max_budget`, `soft_budget`, `team_member_budget`) | team | -| Set on team member | `POST /team/member_add` (`max_budget_in_team`) | team | -| Set on org | `POST /organization/new` (`max_budget`, `soft_budget`, `model_max_budget`) | org | -| Set on customer | `POST /customer/new`, `/customer/update` (`max_budget`, `budget_id`) | customer | -| Set on tag | `POST /tag/new`, `/tag/update` (`max_budget`) | tag mgmt | -| Read budget+spend | `/key/info`, `/user/info`, `/team/info`, `/organization/info`, `/customer/info`, `/budget/info` | per-entity info | - -Endpoint method/shape gotchas verified live: `/organization/delete` is **DELETE** -with `{"organization_ids": [id]}`; `/budget/info` takes `{"budgets": [id]}`; -`model_max_budget` entries use `{"budget_limit", "time_period"}`. - -## 4. Config knobs - -| Setting | Effect | -|---------|--------| -| `litellm.max_budget` | proxy-wide hard cap (global proxy budget) | -| `max_internal_user_budget` / `default_max_internal_user_budget` | default `max_budget` for internal users | -| `internal_user_budget_duration` | default reset duration for internal users | -| `max_end_user_budget` / `max_end_user_budget_id` | default budget for end-users | -| `default_team_params` | default `max_budget` / `budget_duration` / limits for teams | -| `provider_budget_config` (router) | per-provider spend caps + windows | diff --git a/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md b/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md deleted file mode 100644 index 0704ee9b919..00000000000 --- a/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md +++ /dev/null @@ -1,78 +0,0 @@ -# Budget Test Coverage Matrix - -Maps every row of `BUDGET_CODE_MATRIX.md` (what LiteLLM implements) to its tests -and level, then marks the live e2e coverage this suite adds. - -Levels: `unit` mocked (`AsyncMock` on `get_current_spend`/prisma); `router` live -router with fake deployments; `live-e2e` real proxy, real key/team, real requests -until blocked. Status: `covered` / `partial` / `gap`. - -Pre-existing live coverage outside this suite: -- `tests/otel_tests/test_e2e_budgeting.py` - key + team enforcement, budget update. -- `tests/local_testing/test_router_budget_limiter.py` - provider / tag / deployment - budgets at the router. - -This suite (`tests/e2e_tests/budgets/`) adds the missing live coverage and runs -on the shared lifecycle (every entity it creates is deleted on teardown). - ---- - -## Per-entity enforcement - -| Entity | Unit | Pre-existing live | This suite (live) | Status | -|--------|------|-------------------|-------------------|--------| -| API key | `test_budget_reservation.py`, `test_max_budget_limiter.py` | `otel_tests` | `test_budget_enforcement_e2e::test_key_budget_blocks` | **covered** | -| Team | `test_team_budget_limits.py` | `otel_tests` | (org test builds a team) | **covered** | -| Internal user | auth unit tests | - | `test_internal_user_budget_blocks` | **covered (new)** | -| Team member | `test_team_member_budget.py` | - | `test_team_member_budget_blocks` | **covered (new)** | -| End-user / customer | `test_custom_auth_end_user_budget.py` | - | `test_end_user_budget_blocks` | **covered (new)** | -| Organization | `test_organization_budget_enforcement.py` (flagged weak) | - | `test_organization_budget_blocks` | **covered (new)** | -| Tag (proxy-level) | - | router only | `test_tag_budget_e2e::test_tag_budget_blocks_tagged_requests` | **covered (new)** | -| Model-level (`model_max_budget`) | `test_unit_test_max_model_budget_limiter.py` | - | `test_model_max_budget_e2e::test_model_max_budget_isolates_per_model` | **covered (new)** | -| Provider (router) | `test_budget_limiter_hotpath.py` | `test_router_budget_limiter.py` | - | **covered** (router) | -| Global proxy (`litellm.max_budget`) | unit | - | - | **gap** (needs a config-level cap; not key-settable) | - -## Budget mechanisms - -| Mechanism | Unit | This suite (live) | Status | -|-----------|------|-------------------|--------| -| Pre-call reservation | `test_budget_reservation.py` | exercised by every enforcement test | **partial** | -| Soft budget / alerts | `SlackAlerting/test_budget_alert_types.py` | `test_soft_budget_e2e::test_soft_budget_does_not_block` | **covered (new)** (block-vs-alert; the alert side-effect itself stays unit) | -| Budget CRUD | `test_budget_endpoints.py` | `test_budget_crud_e2e` (roundtrip + delete) | **covered (new)** | -| Reset scheduling | `test_proxy_budget_reset.py` | `test_budget_crud_e2e::test_budget_duration_schedules_reset_on_key` | **covered (new)** (scheduling; actual zeroing is time-dependent -> unit) | -| Multi-window budgets | `test_multi_budget_windows.py` | - | **gap** (window setup is fiddly; left to unit for now) | -| Read budget+spend | `test_spend_management_endpoints.py` | `/key/info` asserted in CRUD + enforcement | **partial** | - -## Remaining gaps (intentionally not live-tested) - -- **Global proxy budget** (`litellm.max_budget`): set via proxy config, not a - per-key API, so it needs a dedicated proxy boot with that config rather than a - runtime-created entity. Out of scope for the per-entity suite. -- **Multi-window budgets**: the `budget_limits` list shape and per-window reset are - covered by `test_multi_budget_windows.py` (unit); a live version would need to - wait out a short window to see the reset, which is time-dependent. -- **Soft-budget alert delivery**: whether the Slack/email actually fires is not - observable from the proxy API; unit tests own that. The live test pins the - load-bearing behavior (soft does not block). -- **Reset zeroing after the window elapses**: time-dependent; unit tests own the - reset-job logic. The live test pins that `budget_reset_at` is scheduled. - -## This suite's files - -| File | Covers | -|------|--------| -| `test_budget_enforcement_e2e.py` | key / internal-user / end-user / organization / team-member hard enforcement | -| `test_model_max_budget_e2e.py` | per-model caps isolate by model | -| `test_soft_budget_e2e.py` | soft budget alerts but does not block | -| `test_tag_budget_e2e.py` | proxy-level tag budget blocks tagged requests, spares others | -| `test_budget_crud_e2e.py` | `/budget/*` CRUD roundtrip + delete + `budget_reset_at` scheduling | - -## Pattern + timing - -Create the entity with a tiny `max_budget`, drive spend until a `budget_exceeded` -block. The enforcement helper is two-phase: a fast warmup (key/user/org/member/tag/ -model block within ~2 calls off real-time counters), then a poll across the ~60s -batch-write window (end-user enforcement reads table spend that lags). Skip on a -non-budget error (provider down / key missing); fail if the budget is never -enforced. Chat tests use `gpt-5.5` (the model with a working key on the reference -proxy); swap the literal if your proxy differs. diff --git a/tests/e2e_tests/budgets/budget_client.py b/tests/e2e_tests/budgets/budget_client.py deleted file mode 100644 index 2ca65462d2f..00000000000 --- a/tests/e2e_tests/budgets/budget_client.py +++ /dev/null @@ -1,185 +0,0 @@ -"""Client for budget e2e tests: the shared ProxyClient plus budget-bearing entity -management (user / team / team-member / org / customer / tag / budget-table) and -info reads. - -Over-budget surfaces as a ``budget_exceeded`` error; ``is_budget_block`` detects it -on a CallResult. Create methods return the new id and raise on failure; tests -register the matching delete with ``resources.defer(...)`` for cleanup. -""" - -from typing import Dict, List, Optional - -import requests - -from proxy_client import CallResult, ProxyClient, _auth, proxy_client_kwargs - - -def is_budget_block(result: CallResult) -> bool: - """True if the call was rejected for being over budget (vs a provider error).""" - return not result.ok and "budget_exceeded" in result.body - - -def model_budget(model: str, limit: float, period: str = "30d") -> dict: - """A model_max_budget dict entry: per-model cap with a reset window.""" - return {model: {"budget_limit": limit, "time_period": period}} - - -class BudgetClient(ProxyClient): - def _post(self, path: str, body: Dict[str, object]) -> Dict[str, object]: - resp = requests.post( - f"{self._base_url}{path}", - headers=_auth(self._master_key), - json=body, - timeout=self._request_timeout, - ) - resp.raise_for_status() - return dict(resp.json()) if resp.text else {} - - def _get(self, path: str, params: Dict[str, str]) -> Dict[str, object]: - resp = requests.get( - f"{self._base_url}{path}", - headers=_auth(self._master_key), - params=params, - timeout=self._request_timeout, - ) - resp.raise_for_status() - return dict(resp.json()) - - def _delete(self, path: str, body: Dict[str, object]) -> None: - try: - requests.delete( - f"{self._base_url}{path}", - headers=_auth(self._master_key), - json=body, - timeout=self._request_timeout, - ) - except requests.RequestException: - pass - - def _post_quiet(self, path: str, body: Dict[str, object]) -> None: - try: - requests.post( - f"{self._base_url}{path}", - headers=_auth(self._master_key), - json=body, - timeout=self._request_timeout, - ) - except requests.RequestException: - pass - - # ---- internal user -------------------------------------------------- - - def create_user( - self, *, max_budget: float, budget_duration: Optional[str] = None - ) -> str: - body: Dict[str, object] = {"max_budget": max_budget} - if budget_duration is not None: - body["budget_duration"] = budget_duration - return str(self._post("/user/new", body)["user_id"]) - - def delete_user(self, user_id: str) -> None: - self._post_quiet("/user/delete", {"user_ids": [user_id]}) - - def user_info(self, user_id: str) -> Dict[str, object]: - return self._get("/user/info", {"user_id": user_id}) - - # ---- customer / end-user ------------------------------------------- - - def create_customer(self, customer_id: str, *, max_budget: float) -> str: - self._post("/customer/new", {"user_id": customer_id, "max_budget": max_budget}) - return customer_id - - def customer_info(self, customer_id: str) -> Dict[str, object]: - return self._get("/customer/info", {"end_user_id": customer_id}) - - # ---- organization --------------------------------------------------- - - def create_org(self, *, max_budget: float, alias: str) -> str: - return str( - self._post( - "/organization/new", - {"organization_alias": alias, "max_budget": max_budget}, - )["organization_id"] - ) - - def delete_org(self, org_id: str) -> None: - self._delete("/organization/delete", {"organization_ids": [org_id]}) - - # ---- team ----------------------------------------------------------- - - def create_team( - self, - *, - alias: str, - max_budget: Optional[float] = None, - organization_id: Optional[str] = None, - extra: Optional[Dict[str, object]] = None, - ) -> str: - body: Dict[str, object] = {"team_alias": alias} - if max_budget is not None: - body["max_budget"] = max_budget - if organization_id is not None: - body["organization_id"] = organization_id - if extra: - body.update(extra) - return str(self._post("/team/new", body)["team_id"]) - - def delete_team(self, team_id: str) -> None: - self._post_quiet("/team/delete", {"team_ids": [team_id]}) - - def team_info(self, team_id: str) -> Dict[str, object]: - return self._get("/team/info", {"team_id": team_id}) - - def add_team_member( - self, team_id: str, user_id: str, *, max_budget_in_team: Optional[float] = None - ) -> None: - body: Dict[str, object] = { - "team_id": team_id, - "member": {"role": "user", "user_id": user_id}, - } - if max_budget_in_team is not None: - body["max_budget_in_team"] = max_budget_in_team - self._post("/team/member_add", body) - - # ---- tag ------------------------------------------------------------ - - def create_tag(self, name: str, *, max_budget: float) -> str: - self._post("/tag/new", {"name": name, "max_budget": max_budget}) - return name - - def delete_tag(self, name: str) -> None: - self._post_quiet("/tag/delete", {"name": name}) - - # ---- budget table --------------------------------------------------- - - def create_budget( - self, - *, - max_budget: float, - soft_budget: Optional[float] = None, - budget_duration: Optional[str] = None, - ) -> str: - body: Dict[str, object] = {"max_budget": max_budget} - if soft_budget is not None: - body["soft_budget"] = soft_budget - if budget_duration is not None: - body["budget_duration"] = budget_duration - return str(self._post("/budget/new", body)["budget_id"]) - - def delete_budget(self, budget_id: str) -> None: - self._post_quiet("/budget/delete", {"id": budget_id}) - - def budget_info(self, budget_id: str) -> List[Dict[str, object]]: - resp = requests.post( - f"{self._base_url}/budget/info", - headers=_auth(self._master_key), - json={"budgets": [budget_id]}, - timeout=self._request_timeout, - ) - resp.raise_for_status() - data = resp.json() - return [dict(row) for row in data] if isinstance(data, list) else [] - - -def build_client() -> BudgetClient: - return BudgetClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/budgets/conftest.py b/tests/e2e_tests/budgets/conftest.py deleted file mode 100644 index 12421db91e9..00000000000 --- a/tests/e2e_tests/budgets/conftest.py +++ /dev/null @@ -1,16 +0,0 @@ -"""Budgets suite's `client` fixture. - -The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker -live in the parent tests/e2e_tests/conftest.py. BudgetClient subclasses -ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources` -fixture cleans up keys; tests register entity deletes via `resources.defer(...)`. -""" - -import pytest - -from budget_client import BudgetClient, build_client - - -@pytest.fixture(scope="session") -def client() -> BudgetClient: - return build_client() diff --git a/tests/e2e_tests/budgets/test_budget_crud_e2e.py b/tests/e2e_tests/budgets/test_budget_crud_e2e.py deleted file mode 100644 index 459d019b18e..00000000000 --- a/tests/e2e_tests/budgets/test_budget_crud_e2e.py +++ /dev/null @@ -1,68 +0,0 @@ -"""Live e2e for the budget management surface (no LLM calls, fast). - -Covers the budget-table CRUD round-trip and that `budget_duration` schedules a -`budget_reset_at`. The actual zeroing after the window is time-dependent, so we -assert the reset is *scheduled* (now + duration), not waited out. -""" - -from datetime import datetime, timezone - -import pytest - -from budget_client import BudgetClient -from lifecycle import ResourceManager - -pytestmark = pytest.mark.e2e - - -def _f(value: object) -> float: - return float(value) if value is not None else 0.0 # type: ignore[arg-type] - - -def test_budget_crud_roundtrip( - client: BudgetClient, resources: ResourceManager -) -> None: - budget_id = client.create_budget( - max_budget=12.5, soft_budget=10.0, budget_duration="30d" - ) - resources.defer(lambda: client.delete_budget(budget_id)) - - rows = client.budget_info(budget_id) - assert rows, f"/budget/info returned nothing for {budget_id}" - row = rows[0] - assert _f(row.get("max_budget")) == 12.5 - assert _f(row.get("soft_budget")) == 10.0 - assert row.get("budget_reset_at"), "budget_duration did not schedule a reset" - - # Attach the budget to a key and confirm the key reflects it. - key = client.generate_key(extra_params={"budget_id": budget_id}) - resources.defer(lambda: client.delete_key(key)) - info = client.key_info(key) - linked = info.get("litellm_budget_table") or {} - assert ( - info.get("budget_id") == budget_id or _f(linked.get("max_budget")) == 12.5 - ), f"key does not reflect attached budget: {info.get('budget_id')}, {linked}" - - -def test_budget_delete_removes_it( - client: BudgetClient, resources: ResourceManager -) -> None: - budget_id = client.create_budget(max_budget=1.0) - client.delete_budget(budget_id) - assert client.budget_info(budget_id) == [], "budget still present after delete" - - -def test_budget_duration_schedules_reset_on_key( - client: BudgetClient, resources: ResourceManager -) -> None: - key = client.generate_key( - max_budget=10.0, extra_params={"budget_duration": "30d"} - ) - resources.defer(lambda: client.delete_key(key)) - - reset_at = client.key_info(key).get("budget_reset_at") - assert reset_at, "budget_duration did not set budget_reset_at on the key" - - reset_dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00")) - days_out = (reset_dt - datetime.now(timezone.utc)).total_seconds() / 86400 - assert 28 < days_out < 32, f"reset ~30d expected, got {days_out:.1f}d out" diff --git a/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py b/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py deleted file mode 100644 index 591776ca22e..00000000000 --- a/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py +++ /dev/null @@ -1,119 +0,0 @@ -"""Live e2e: a tiny max_budget on an entity actually blocks requests. - -Mirrors tests/otel_tests/test_e2e_budgeting.py (call until `budget_exceeded`) but -covers the entities that had NO live coverage - internal user, end-user, -organization, team member - and runs in this suite so the shared `resources` -teardown deletes every entity created. See BUDGET_TEST_COVERAGE_MATRIX.md. - -Skip on environment, fail on behavior: a non-budget error (provider down) skips; -if calls never get blocked, budget enforcement is broken -> fail. -""" - -import time - -import pytest - -from budget_client import BudgetClient, is_budget_block -from lifecycle import ResourceManager -from proxy_client import require_successful_call, unique_marker - -pytestmark = pytest.mark.e2e - -TINY_BUDGET = 1e-6 # one real call's spend exceeds this, so the block lands fast - - -def _assert_budget_blocks( - client: BudgetClient, key: str, model: str, *, user: str = "" -) -> None: - """Drive spend over the entity's budget and assert a call is blocked. - - Two-phase so it's robust across entities: key/user/org enforce off real-time - reservation counters (block within a couple calls), but end-user enforcement - reads the table spend that only updates on the ~60s batch write - so after a - warmup we poll for the block across that window. Skip on a non-budget error; - fail if the budget is never enforced. - """ - extra: dict = {"max_tokens": 16} - if user: - extra["user"] = user - - def _call(label: str): - result = client.chat(key, model, f"{label} {unique_marker()}", extra_body=extra) - if not result.ok and not is_budget_block(result): - require_successful_call(result) # non-budget error -> skip - return result - - for _ in range(2): # warmup: incur spend (fast entities block here) - if is_budget_block(_call("warmup")): - return - time.sleep(0.5) - - deadline = time.monotonic() + 100 # cover the batch-write propagation window - while time.monotonic() < deadline: - if is_budget_block(_call("probe")): - return - time.sleep(5) - pytest.fail("budget not enforced within 100s") - - -def test_key_budget_blocks(client: BudgetClient, resources: ResourceManager) -> None: - key = client.generate_key(max_budget=TINY_BUDGET) - resources.defer(lambda: client.delete_key(key)) - _assert_budget_blocks(client, key, "gpt-5.5") - - -def test_internal_user_budget_blocks( - client: BudgetClient, resources: ResourceManager -) -> None: - user_id = client.create_user(max_budget=TINY_BUDGET) - resources.defer(lambda: client.delete_user(user_id)) - # personal key (no team) -> the user budget governs - key = client.generate_key(extra_params={"user_id": user_id}) - resources.defer(lambda: client.delete_key(key)) - _assert_budget_blocks(client, key, "gpt-5.5") - - -def test_end_user_budget_blocks( - client: BudgetClient, scoped_key: str, resources: ResourceManager -) -> None: - customer = f"e2e-budget-cust-{unique_marker()}" - client.create_customer(customer, max_budget=TINY_BUDGET) - resources.defer(lambda: client.delete_customers([customer])) - _assert_budget_blocks(client, scoped_key, "gpt-5.5", user=customer) - - -def test_organization_budget_blocks( - client: BudgetClient, resources: ResourceManager -) -> None: - # Org carries the tiny budget; the team under it has none, so a block here is - # org-level enforcement (the historically weak link). - org_id = client.create_org( - max_budget=TINY_BUDGET, alias=f"e2e-budget-org-{unique_marker()}" - ) - resources.defer(lambda: client.delete_org(org_id)) - team_id = client.create_team( - alias=f"e2e-budget-team-{unique_marker()}", organization_id=org_id - ) - resources.defer(lambda: client.delete_team(team_id)) - key = client.generate_key(extra_params={"team_id": team_id}) - resources.defer(lambda: client.delete_key(key)) - _assert_budget_blocks(client, key, "gpt-5.5") - - -def test_team_member_budget_blocks( - client: BudgetClient, resources: ResourceManager -) -> None: - # Member's per-team budget is tiny while the team itself has a large budget, - # so a block proves member-level (not team-level) enforcement. - team_id = client.create_team( - alias=f"e2e-budget-team-{unique_marker()}", max_budget=100.0 - ) - resources.defer(lambda: client.delete_team(team_id)) - user_id = client.create_user(max_budget=100.0) - resources.defer(lambda: client.delete_user(user_id)) - client.add_team_member(team_id, user_id, max_budget_in_team=TINY_BUDGET) - key = client.generate_key( - extra_params={"team_id": team_id, "user_id": user_id} - ) - resources.defer(lambda: client.delete_key(key)) - _assert_budget_blocks(client, key, "gpt-5.5") diff --git a/tests/e2e_tests/budgets/test_model_max_budget_e2e.py b/tests/e2e_tests/budgets/test_model_max_budget_e2e.py deleted file mode 100644 index 7db8c3fec2e..00000000000 --- a/tests/e2e_tests/budgets/test_model_max_budget_e2e.py +++ /dev/null @@ -1,60 +0,0 @@ -"""Live e2e: per-model budgets (`model_max_budget`) isolate by model. - -A key caps one model tiny and leaves another generous. Exhausting the capped -model must block *that* model while the other still works - proving the per-model -cap is enforced independently, not as a key-wide budget. Closes the -model_max_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. -""" - -import time - -import pytest - -from budget_client import BudgetClient, is_budget_block, model_budget -from lifecycle import ResourceManager -from proxy_client import require_successful_call, unique_marker - -pytestmark = pytest.mark.e2e - -CAPPED_MODEL = "gpt-5.5" -FREE_MODEL = "claude-haiku-4-5" - - -def _call(client: BudgetClient, key: str, model: str): - result = client.chat( - key, model, f"hi {unique_marker()}", extra_body={"max_tokens": 16} - ) - if not result.ok and not is_budget_block(result): - require_successful_call(result) # non-budget error -> skip - return result - - -def test_model_max_budget_isolates_per_model( - client: BudgetClient, resources: ResourceManager -) -> None: - key = client.generate_key( - extra_params={ - "model_max_budget": { - **model_budget(CAPPED_MODEL, 1e-6), - **model_budget(FREE_MODEL, 1000.0), - } - } - ) - resources.defer(lambda: client.delete_key(key)) - - # Exhaust the capped model. - blocked = False - deadline = time.monotonic() + 60 - while time.monotonic() < deadline: - if is_budget_block(_call(client, key, CAPPED_MODEL)): - blocked = True - break - time.sleep(1) - assert blocked, f"{CAPPED_MODEL} per-model budget never enforced" - - # The other model shares the key but has its own (large) cap -> still works. - other = _call(client, key, FREE_MODEL) - require_successful_call(other) # skips if claude key absent; never a budget block - assert not is_budget_block(other), ( - f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" - ) diff --git a/tests/e2e_tests/budgets/test_soft_budget_e2e.py b/tests/e2e_tests/budgets/test_soft_budget_e2e.py deleted file mode 100644 index cd957d9dd90..00000000000 --- a/tests/e2e_tests/budgets/test_soft_budget_e2e.py +++ /dev/null @@ -1,36 +0,0 @@ -"""Live e2e: soft_budget alerts but does NOT block. - -A key with a tiny `soft_budget` well under a large `max_budget`: spend crosses the -soft threshold within a couple calls, but requests keep succeeding (soft budget is -advisory). Closes the soft_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. The alert -side-effect (Slack/email) is not observable from the proxy API, so we assert the -load-bearing behavior: soft != block. -""" - -import pytest - -from budget_client import BudgetClient, is_budget_block -from lifecycle import ResourceManager -from proxy_client import require_successful_call, unique_marker - -pytestmark = pytest.mark.e2e - - -def test_soft_budget_does_not_block( - client: BudgetClient, resources: ResourceManager -) -> None: - # soft far below max: spend crosses soft immediately, stays under max. - key = client.generate_key( - max_budget=1000.0, extra_params={"soft_budget": 1e-9} - ) - resources.defer(lambda: client.delete_key(key)) - - for _ in range(3): - result = client.chat( - key, "gpt-5.5", f"hi {unique_marker()}", extra_body={"max_tokens": 16} - ) - require_successful_call(result) # skip if provider unavailable - assert not is_budget_block(result), ( - "soft_budget blocked a request; it must alert only, not block " - f"(body={result.body[:200]})" - ) diff --git a/tests/e2e_tests/budgets/test_tag_budget_e2e.py b/tests/e2e_tests/budgets/test_tag_budget_e2e.py deleted file mode 100644 index 5dc15720664..00000000000 --- a/tests/e2e_tests/budgets/test_tag_budget_e2e.py +++ /dev/null @@ -1,58 +0,0 @@ -"""Live e2e: proxy-level tag budgets block tagged requests. - -A tag with a tiny budget: requests carrying that tag get blocked once the tag's -spend is exceeded, while a request with a different tag (no budget) still works. -Closes the proxy-level tag-budget gap in BUDGET_TEST_COVERAGE_MATRIX.md (today -only router-level tag budgets are tested). -""" - -import time - -import pytest - -from budget_client import BudgetClient, is_budget_block -from lifecycle import ResourceManager -from proxy_client import require_successful_call, unique_marker - -pytestmark = pytest.mark.e2e - -TINY_BUDGET = 1e-6 - - -def _tagged_call(client: BudgetClient, key: str, tag: str): - result = client.chat( - key, - "gpt-5.5", - f"hi {unique_marker()}", - metadata={"tags": [tag]}, - extra_body={"max_tokens": 16}, - ) - if not result.ok and not is_budget_block(result): - require_successful_call(result) # non-budget error -> skip - return result - - -def test_tag_budget_blocks_tagged_requests( - client: BudgetClient, scoped_key: str, resources: ResourceManager -) -> None: - budgeted_tag = f"e2e-budget-tag-{unique_marker()}" - client.create_tag(budgeted_tag, max_budget=TINY_BUDGET) - resources.defer(lambda: client.delete_tag(budgeted_tag)) - - # Requests under the budgeted tag get blocked once its spend is exceeded. - blocked = False - deadline = time.monotonic() + 60 - while time.monotonic() < deadline: - if is_budget_block(_tagged_call(client, scoped_key, budgeted_tag)): - blocked = True - break - time.sleep(1) - assert blocked, f"tag budget for {budgeted_tag!r} never enforced" - - # A request with an unbudgeted tag on the same key is unaffected. - free_tag = f"e2e-free-tag-{unique_marker()}" - other = _tagged_call(client, scoped_key, free_tag) - require_successful_call(other) - assert not is_budget_block(other), ( - f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget" - ) diff --git a/tests/e2e_tests/conftest.py b/tests/e2e_tests/conftest.py deleted file mode 100644 index 4f73bf54786..00000000000 --- a/tests/e2e_tests/conftest.py +++ /dev/null @@ -1,54 +0,0 @@ -"""Shared fixtures for all live e2e suites under tests/e2e_tests/. - -Design rule: skip on environment, fail on behavior. If the proxy is unreachable -the whole session skips; once a request reaches the proxy, behavior is asserted. - -Lifecycle: the `resources` fixture maps the init -> run -> teardown contract -(lifecycle.E2ECase) onto pytest - setup is init(), the test body is run(), and -teardown deletes every resource the test created on the long-lived proxy. - -Each suite provides its own `client` fixture (a lifecycle.ResourceClient); these -shared fixtures build on it. -""" - -from typing import Iterator - -import pytest -import requests - -from e2e_config import PROXY_BASE_URL -from lifecycle import ResourceClient, ResourceManager - - -def pytest_configure(config: pytest.Config) -> None: - config.addinivalue_line( - "markers", - "e2e: live test that requires a running proxy and real provider keys", - ) - - -@pytest.fixture(scope="session", autouse=True) -def _require_live_proxy() -> None: - """Skip the entire session unless a proxy answers its liveness probe.""" - try: - resp = requests.get(f"{PROXY_BASE_URL}/health/liveliness", timeout=5) - except requests.RequestException as exc: - pytest.skip(f"No live proxy at {PROXY_BASE_URL}: {exc}") - return - if resp.status_code >= 500: - pytest.skip(f"Proxy at {PROXY_BASE_URL} returned {resp.status_code}") - - -@pytest.fixture -def resources(client: ResourceClient) -> Iterator[ResourceManager]: - """init -> run -> teardown: create a manager, run the test, release resources.""" - manager = ResourceManager(client=client) - manager.init() - yield manager - manager.teardown() - - -@pytest.fixture -def scoped_key(resources: ResourceManager) -> str: - """A fresh all-models key per test, auto-deleted by the resources teardown.""" - return resources.key() diff --git a/tests/e2e_tests/docker-compose.yml b/tests/e2e_tests/docker-compose.yml deleted file mode 100644 index ab237f01d15..00000000000 --- a/tests/e2e_tests/docker-compose.yml +++ /dev/null @@ -1,59 +0,0 @@ -# Local e2e proxy, built from the litellm_internal_staging branch. -# -# The image `litellm-staging:local` is built from a clean checkout of -# litellm_internal_staging (the working tree's uv.lock has merge markers, so -# build from a worktree, not the dirty tree): -# -# git worktree add /tmp/litellm-staging litellm_internal_staging -# docker build -t litellm-staging:local /tmp/litellm-staging -# -# Then provide provider keys (copy .env.example -> .env, fill in keys) and run: -# -# docker compose -f tests/e2e_tests/docker-compose.yml up -d -# -# The proxy comes up on http://localhost:4000 with master key sk-1234, which is -# what the e2e suites default to. Point the suites at it: -# -# LITELLM_MASTER_KEY=sk-1234 LITELLM_PROXY_URL=http://localhost:4000 \ -# .venv/bin/python -m pytest tests/e2e_tests/ -v - -services: - litellm: - image: litellm-staging:local - ports: - - "4000:4000" - command: ["--config", "/app/config.yaml", "--port", "4000"] - volumes: - - ./docker-config.yaml:/app/config.yaml - environment: - DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm" - LITELLM_MASTER_KEY: "sk-1234" - STORE_MODEL_IN_DB: "True" - env_file: - # Provider keys for the LLM-calling suites. Boots without it (those tests - # skip); copy .env.example -> .env and `docker compose up -d` to enable. - - path: .env - required: false - depends_on: - - db - - redis - - db: - image: postgres:17 - environment: - POSTGRES_DB: litellm - POSTGRES_USER: llmproxy - POSTGRES_PASSWORD: dbpassword9090 - ports: - - "5432:5432" - volumes: - - postgres_data:/var/lib/postgresql/data - - redis: - image: redis:7-alpine - ports: - - "6379:6379" - -volumes: - postgres_data: - driver: local diff --git a/tests/e2e_tests/docker-config.yaml b/tests/e2e_tests/docker-config.yaml deleted file mode 100644 index f0869c3c06b..00000000000 --- a/tests/e2e_tests/docker-config.yaml +++ /dev/null @@ -1,39 +0,0 @@ -# Config for the local docker e2e proxy (see docker-compose.yml). -# Minimal on purpose: just the models the e2e suites use + spend tracking + -# a redis cache (for the cache-hit spend test). Provider keys come from .env. -# Passthrough endpoints (/gemini, /anthropic) use GEMINI_API_KEY / ANTHROPIC_API_KEY -# from the environment directly, so no model_list entry is needed for them. - -general_settings: - master_key: os.environ/LITELLM_MASTER_KEY - database_url: os.environ/DATABASE_URL - store_prompts_in_spend_logs: true - -litellm_settings: - drop_params: true - cache: true - cache_params: - type: redis - host: redis - port: 6379 - -model_list: - - model_name: gpt-5.5 - litellm_params: - model: openai/gpt-5.5 - api_key: os.environ/OPENAI_API_KEY - - - model_name: claude-haiku-4-5 - litellm_params: - model: anthropic/claude-haiku-4-5 - api_key: os.environ/ANTHROPIC_API_KEY - - - model_name: gemini-2.5-flash - litellm_params: - model: gemini/gemini-2.5-flash - api_key: os.environ/GEMINI_API_KEY - - - model_name: openai-text-embedding-3-small - litellm_params: - model: openai/text-embedding-3-small - api_key: os.environ/OPENAI_API_KEY diff --git a/tests/e2e_tests/e2e_config.py b/tests/e2e_tests/e2e_config.py deleted file mode 100644 index 56834d9abb0..00000000000 --- a/tests/e2e_tests/e2e_config.py +++ /dev/null @@ -1,16 +0,0 @@ -"""Generic configuration for live e2e tests against a running LiteLLM proxy. - -Shared by every e2e suite under tests/e2e_tests/. Values come from the -environment so the same tests run against localhost or a deployed proxy. -""" - -import os - -PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/") -MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234") - -# Writes on the proxy are eventually consistent (e.g. spend rows flush on -# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once. -POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120")) -POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5")) -REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60")) diff --git a/tests/e2e_tests/lifecycle.py b/tests/e2e_tests/lifecycle.py deleted file mode 100644 index 8130e388bff..00000000000 --- a/tests/e2e_tests/lifecycle.py +++ /dev/null @@ -1,87 +0,0 @@ -"""Lifecycle contract and resource cleanup for stateful e2e tests. - -Shared by every e2e suite under tests/e2e_tests/. The proxy under test is -long-lived and never reset between tests, so anything a test creates (keys, -customers, teams, orgs, users, guardrails, budgets, ...) persists unless -explicitly deleted. Every check follows an init -> run -> teardown lifecycle; -teardown releases each resource init() created, even when run() raises. - -In pytest terms (see conftest.py): the `resources` fixture's setup is init(), -the test body is run(), and the fixture's teardown is teardown(). -""" - -from dataclasses import dataclass, field -from typing import Callable, List, Protocol, runtime_checkable - - -@runtime_checkable -class E2ECase(Protocol): - """A stateful e2e check run against a long-lived proxy. - - init() acquires resources, run() exercises behaviour and asserts, teardown() - releases everything init() created. teardown() must run even if run() raises. - """ - - def init(self) -> None: ... - - def run(self) -> None: ... - - def teardown(self) -> None: ... - - -@runtime_checkable -class ResourceClient(Protocol): - """Proxy operations the convenience creators use. Resource types without a - creator here are handled generically via ResourceManager.defer().""" - - def generate_key(self, *, models: List[str]) -> str: ... - - def delete_key(self, key: str) -> None: ... - - def delete_customers(self, user_ids: List[str]) -> None: ... - - -@dataclass -class ResourceManager: - """Registry of teardown actions for resources a test creates on the stateful - proxy. - - Not limited to any resource type: register a cleanup with ``defer()`` for a - key, customer, team, org, user, guardrail, budget, MCP server - anything with - a delete. The two most common resources have sugar (``key``, ``customer``); - everything else is ``resources.defer(lambda: client.delete_team(team_id))``. - - Cleanups run LIFO (so a resource is removed before whatever it depends on) and - best-effort (one failing cleanup never blocks the rest). - """ - - client: ResourceClient - _cleanups: List[Callable[[], None]] = field( - default_factory=list - ) # mutable-ok: append-only teardown registry - - def init(self) -> None: - """No global setup needed today; present for lifecycle symmetry.""" - return None - - def defer(self, cleanup: Callable[[], None]) -> None: - """Register a teardown action for any resource the test just created.""" - self._cleanups.append(cleanup) - - def key(self) -> str: - """Create an all-models virtual key; delete it on teardown.""" - key = self.client.generate_key(models=[]) - self.defer(lambda: self.client.delete_key(key)) - return key - - def customer(self, customer_id: str) -> str: - """Track an end-user id (from the `user` param); delete it on teardown.""" - self.defer(lambda: self.client.delete_customers([customer_id])) - return customer_id - - def teardown(self) -> None: - for cleanup in reversed(self._cleanups): - try: - cleanup() - except Exception: - pass # best-effort: a failed cleanup must not block the rest diff --git a/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md b/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md deleted file mode 100644 index 5e4a448857f..00000000000 --- a/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md +++ /dev/null @@ -1,84 +0,0 @@ -# LLM Translation Test Coverage Matrix - -Scope: the proxy's two translation surfaces, end to end against a live proxy. - -1. **Passthrough** - the client speaks the provider's NATIVE API (Gemini - `generateContent`, Anthropic `/v1/messages`); the proxy forwards it and still - logs a costed `SpendLogs` row (`call_type="pass_through_endpoint"`). Routes: - `/gemini`, `/anthropic`, `/vertex_ai`, `/openai`, `/bedrock`, `/cohere`, - `/mistral`, `/vllm`. -2. **Non-passthrough** - the client speaks OpenAI format - (`/chat/completions`, `/embeddings`); litellm translates to/from the provider. - -The two axes that must work in production for each: **passthrough vs -non-passthrough** and **streaming vs non-streaming**, with **cost logged** and -**tool calls** working in every cell. - -Companion: live suite `test_passthrough_e2e.py` (this directory). The -non-passthrough chat/embedding cells are exercised by `../spend_tracking/`. - -Levels: `live` real provider + proxy + SpendLogs row; `unit` mocked. -Status: `covered` / `partial` / `gap`. - ---- - -## Passthrough endpoints (native provider format) - -| Provider | Non-streaming | Streaming | Tool calls | Cost logged | Status | -|----------|---------------|-----------|------------|-------------|--------| -| Gemini (`/gemini/v1beta/models/{m}:generateContent` / `:streamGenerateContent`) | live | live | live | live | **covered** | -| Anthropic (`/anthropic/v1/messages`) | live | live | live | live | **covered** | -| Vertex AI (`/vertex_ai/...`) | - | - | - | - | gap (gcloud auth) | -| OpenAI / Bedrock / Cohere / Mistral / VLLM | - | - | - | - | gap | - -Each covered cell asserts: `call_type == "pass_through_endpoint"`, `spend > 0`, -`status == "success"`, correct `custom_llm_provider`/`model`, row correlated by the -`x-litellm-call-id` header. Gemini non-streaming also pins `request_tags` -propagation; streaming pins `chunks > 0` then a costed row; tool tests assert the -provider emitted a tool call (`functionCall` / `tool_use`) and it was costed. - -Cost on passthrough is computed in the success handler by transforming the native -response to a `ModelResponse` and calling `litellm.completion_cost()`; for -streaming, chunks are buffered and costed after the stream ends. This is the path -most likely to silently break and the one a mock can't prove works. - -## Non-passthrough endpoints (OpenAI-compatible translation) - -| Modality | Non-streaming | Streaming | Tool calls | Cost logged | Status | -|----------|---------------|-----------|------------|-------------|--------| -| Chat | live (spend suite) | live (spend suite) | gap | live | partial | -| Embeddings | live (spend suite) | n/a | n/a | live | covered | -| Responses / image / audio / rerank / realtime | - | - | - | - | gap | - -## This suite's files - -| Test | Cell | -|------|------| -| `test_gemini_passthrough_nonstreaming_logs_cost` | gemini native, non-stream, cost + tags | -| `test_gemini_passthrough_streaming_logs_cost` | gemini native, stream, cost | -| `test_gemini_passthrough_tool_call_logs_cost` | gemini native, tool call, cost | -| `test_anthropic_passthrough_nonstreaming_logs_cost` | anthropic native, non-stream, cost | -| `test_anthropic_passthrough_streaming_logs_cost` | anthropic native, stream, cost | -| `test_anthropic_passthrough_tool_call_logs_cost` | anthropic native, tool call, cost | - -## Gaps - -- Vertex / OpenAI / Bedrock / Cohere passthrough (same shape; add once the - provider credential is configured; Vertex is closest - route exists, auth stale). -- Non-passthrough tool calls over `/chat/completions` end to end with cost. -- Image / audio / rerank / responses / realtime translation + cost. -- Streaming cost-injection (`include_cost_in_streaming_usage`); passthrough on - client disconnect (partial-usage logging). - -## Adding a provider/modality - -Extend `PassthroughClient` with the native call (it inherits keys, cleanup, and -SpendLogs polling from `ProxyClient`), then add a test that calls it, -`require_successful_call(result)`, and `_costed_row(...)`. - -## Timing - -Passthrough spend is logged asynchronously after the response and lands on the -`proxy_batch_write_at` (~60s) cycle, so cost assertions poll -`/spend/logs?request_id=` to a deadline. Streaming cost is only -known after the stream is fully consumed. diff --git a/tests/e2e_tests/llm_translation/conftest.py b/tests/e2e_tests/llm_translation/conftest.py deleted file mode 100644 index 58f2b3e4dc1..00000000000 --- a/tests/e2e_tests/llm_translation/conftest.py +++ /dev/null @@ -1,16 +0,0 @@ -"""LLM-translation suite's `client` fixture. - -The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker -live in the parent tests/e2e_tests/conftest.py. PassthroughClient subclasses -ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources` -fixture cleans up keys this suite creates. -""" - -import pytest - -from passthrough_client import PassthroughClient, build_client - - -@pytest.fixture(scope="session") -def client() -> PassthroughClient: - return build_client() diff --git a/tests/e2e_tests/llm_translation/passthrough_client.py b/tests/e2e_tests/llm_translation/passthrough_client.py deleted file mode 100644 index 458f31edff8..00000000000 --- a/tests/e2e_tests/llm_translation/passthrough_client.py +++ /dev/null @@ -1,136 +0,0 @@ -"""Client for LLM-translation e2e tests over the proxy's passthrough endpoints. - -Extends the shared ProxyClient with native provider passthrough calls. A -passthrough request is sent in the PROVIDER's native format (Gemini -generateContent, Anthropic /v1/messages) to the proxy, which forwards it to the -provider and still logs a SpendLogs row (call_type="pass_through_endpoint"). The -litellm virtual key is passed as the provider key; the proxy swaps in the real -env credential. SpendLogs.request_id == the x-litellm-call-id response header. -""" - -from dataclasses import dataclass -from typing import List, Optional - -import requests - -from proxy_client import ProxyClient, proxy_client_kwargs - - -@dataclass(frozen=True, slots=True) -class PassthroughResult: - """Outcome of a native passthrough call. ``call_id`` correlates to the row.""" - - status_code: int - call_id: Optional[str] # x-litellm-call-id -> SpendLogs.request_id - body: str - chunks: int = 0 # number of streamed events (0 for non-streaming) - - @property - def ok(self) -> bool: - return 200 <= self.status_code < 300 - - -def _tag_header(tags: Optional[List[str]]) -> dict: - return {"tags": ",".join(tags)} if tags else {} - - -class PassthroughClient(ProxyClient): - # ---- Gemini native passthrough (/gemini/v1beta/...) ----------------- - - def gemini_generate( - self, - key: str, - model: str, - text: str, - *, - tools: Optional[list] = None, - tags: Optional[List[str]] = None, - ) -> PassthroughResult: - body: dict = {"contents": [{"role": "user", "parts": [{"text": text}]}]} - if tools is not None: - body["tools"] = tools - headers = { - "x-goog-api-key": key, - "Content-Type": "application/json", - **_tag_header(tags), - } - resp = requests.post( - f"{self._base_url}/gemini/v1beta/models/{model}:generateContent", - headers=headers, - json=body, - timeout=self._request_timeout, - ) - return PassthroughResult( - resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text - ) - - def gemini_stream( - self, key: str, model: str, text: str, *, tags: Optional[List[str]] = None - ) -> PassthroughResult: - headers = { - "x-goog-api-key": key, - "Content-Type": "application/json", - **_tag_header(tags), - } - resp = requests.post( - f"{self._base_url}/gemini/v1beta/models/{model}:streamGenerateContent", - headers=headers, - params={"alt": "sse"}, - json={"contents": [{"role": "user", "parts": [{"text": text}]}]}, - stream=True, - timeout=self._request_timeout, - ) - call_id = resp.headers.get("x-litellm-call-id") - if not (200 <= resp.status_code < 300): - return PassthroughResult(resp.status_code, call_id, resp.text) - chunks = sum(1 for line in resp.iter_lines() if line) - return PassthroughResult(resp.status_code, call_id, "", chunks) - - # ---- Anthropic native passthrough (/anthropic/v1/messages) ---------- - - def anthropic_message( - self, - key: str, - model: str, - text: str, - *, - max_tokens: int = 64, - tools: Optional[list] = None, - stream: bool = False, - tags: Optional[List[str]] = None, - ) -> PassthroughResult: - body: dict = { - "model": model, - "max_tokens": max_tokens, - "messages": [{"role": "user", "content": text}], - } - if tools is not None: - body["tools"] = tools - if stream: - body["stream"] = True - headers = { - "x-api-key": key, - "anthropic-version": "2023-06-01", - "Content-Type": "application/json", - **_tag_header(tags), - } - url = f"{self._base_url}/anthropic/v1/messages" - if not stream: - resp = requests.post( - url, headers=headers, json=body, timeout=self._request_timeout - ) - return PassthroughResult( - resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text - ) - resp = requests.post( - url, headers=headers, json=body, stream=True, timeout=self._request_timeout - ) - call_id = resp.headers.get("x-litellm-call-id") - if not (200 <= resp.status_code < 300): - return PassthroughResult(resp.status_code, call_id, resp.text) - chunks = sum(1 for line in resp.iter_lines() if line) - return PassthroughResult(resp.status_code, call_id, "", chunks) - - -def build_client() -> PassthroughClient: - return PassthroughClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/llm_translation/test_passthrough_e2e.py b/tests/e2e_tests/llm_translation/test_passthrough_e2e.py deleted file mode 100644 index b7250898200..00000000000 --- a/tests/e2e_tests/llm_translation/test_passthrough_e2e.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Live e2e for LLM-translation passthrough endpoints. - -Each test sends a NATIVE provider request through the proxy's passthrough route -and verifies the proxy still logged a costed SpendLogs row -(call_type="pass_through_endpoint"), correlated by the x-litellm-call-id header. - -Covered: gemini ("gemini-2.5-flash") + anthropic ("claude-haiku-4-5"), streaming + -non-streaming, plus native tool calls. See LLM_TRANSLATION_COVERAGE_MATRIX.md. - -Skip on environment, fail on behavior: a passthrough call returning non-2xx -(provider key missing) skips; once it returns 2xx, a missing/zero-cost row fails. -""" - -import pytest - -from passthrough_client import PassthroughClient, PassthroughResult -from proxy_client import SpendLogRow, require_successful_call, unique_marker - -pytestmark = pytest.mark.e2e - - -def _f(value: object) -> float: - return float(value) if value is not None else 0.0 # type: ignore[arg-type] - - -def _s(value: object) -> str: - return str(value) if value is not None else "" - - -def _costed_row(client: PassthroughClient, result: PassthroughResult) -> SpendLogRow: - """The passthrough call's logged row, polled until it carries a cost. - - Asserts (not skips) that a 2xx passthrough call produced a costed row - the - whole point of passthrough spend tracking. - """ - assert result.call_id, "passthrough response had no x-litellm-call-id header" - rows = client.poll_logs_for_request_id( - result.call_id, - predicate=lambda rs: _f(rs[0].get("spend")) > 0, - ) - assert rows, f"no SpendLogs row for passthrough call_id {result.call_id}" - row = rows[0] - assert _s(row.get("call_type")) == "pass_through_endpoint" - assert _f(row.get("spend")) > 0, f"passthrough call was not costed: {row}" - assert _s(row.get("status")) == "success" - return row - - -# ---- Gemini passthrough ------------------------------------------------ - - -def test_gemini_passthrough_nonstreaming_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - tag = f"e2e-passthrough-{unique_marker()}" - result = client.gemini_generate( - scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"] - ) - require_successful_call(result) - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "gemini" - assert "gemini" in _s(row.get("model")) - assert tag in _s(row.get("request_tags")), f"tags not logged: {row.get('request_tags')}" - - -def test_gemini_passthrough_streaming_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five") - require_successful_call(result) - assert result.chunks > 0, "streaming passthrough produced no events" - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "gemini" - - -def test_gemini_passthrough_tool_call_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - result = client.gemini_generate( - scoped_key, - "gemini-2.5-flash", - "What is the weather in Paris? Use the get_weather tool.", - tools=[ - { - "functionDeclarations": [ - { - "name": "get_weather", - "description": "Get the weather for a city", - "parameters": { - "type": "object", - "properties": {"city": {"type": "string"}}, - "required": ["city"], - }, - } - ] - } - ], - ) - require_successful_call(result) - assert "functionCall" in result.body, "gemini did not emit a tool call" - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "gemini" - - -# ---- Anthropic passthrough --------------------------------------------- - - -def test_anthropic_passthrough_nonstreaming_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello") - require_successful_call(result) - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "anthropic" - assert "claude" in _s(row.get("model")) - - -def test_anthropic_passthrough_streaming_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - result = client.anthropic_message( - scoped_key, "claude-haiku-4-5", "Count to five", stream=True - ) - require_successful_call(result) - assert result.chunks > 0, "streaming passthrough produced no events" - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "anthropic" - - -def test_anthropic_passthrough_tool_call_logs_cost( - client: PassthroughClient, scoped_key: str -) -> None: - result = client.anthropic_message( - scoped_key, - "claude-haiku-4-5", - "What is the weather in Paris? Use the get_weather tool.", - tools=[ - { - "name": "get_weather", - "description": "Get the weather for a city", - "input_schema": { - "type": "object", - "properties": {"city": {"type": "string"}}, - "required": ["city"], - }, - } - ], - ) - require_successful_call(result) - assert "tool_use" in result.body, "anthropic did not emit a tool call" - - row = _costed_row(client, result) - assert _s(row.get("custom_llm_provider")) == "anthropic" diff --git a/tests/e2e_tests/proxy_client.py b/tests/e2e_tests/proxy_client.py deleted file mode 100644 index cadbb9d5268..00000000000 --- a/tests/e2e_tests/proxy_client.py +++ /dev/null @@ -1,371 +0,0 @@ -"""Generic HTTP client for live e2e tests against a running LiteLLM proxy. - -Shared by every e2e suite under tests/e2e_tests/. Covers the proxy operations any -suite needs: key/customer management (so the shared ResourceManager can clean up), -OpenAI-compatible calls, route probing, and SpendLogs read-back. Suite-specific -clients subclass ProxyClient (see spend_tracking/, llm_translation/, budgets/). - -Talks to the proxy over real HTTP so every test sees what a real client sees: -the x-litellm-call-id header, the response body, and the rows the proxy writes. -Writes are eventually consistent (proxy_batch_write_at ~60s), so read-backs poll -to a deadline rather than sleeping once. -""" - -import json -import time -import uuid -from dataclasses import dataclass -from typing import Callable, Dict, List, Optional, Protocol, runtime_checkable - -import pytest -import requests - -from e2e_config import ( - MASTER_KEY, - POLL_INTERVAL, - POLL_TIMEOUT, - PROXY_BASE_URL, - REQUEST_TIMEOUT, -) - -SpendLogRow = Dict[str, object] - - -@runtime_checkable -class CallOutcome(Protocol): - """Anything with an HTTP status and body that can pass the skip/fail boundary.""" - - status_code: int - body: str - - @property - def ok(self) -> bool: ... - - -@dataclass(frozen=True, slots=True) -class ProbeResult: - """Outcome of a single route probe: enough to see *why* it (mis)behaved.""" - - url: str - status_code: int - body: str - - @property - def healthy(self) -> bool: - # Route exists (not 404), handler did not crash (not 5xx), request - # completed (not -1). A 4xx (missing params/auth) still means it ran. - return 200 <= self.status_code < 500 and self.status_code != 404 - - def __str__(self) -> str: - return f"GET {self.url} -> {self.status_code}\n{self.body[:600]}" - - -@dataclass(frozen=True, slots=True) -class CallResult: - """Outcome of a single OpenAI-compatible call made through the proxy.""" - - status_code: int - call_id: Optional[str] # x-litellm-call-id response header - response_id: Optional[str] # body "id"; SpendLogs.request_id is derived from this - response_cost_header: Optional[str] # x-litellm-response-cost header - body: str - content: Optional[str] - - @property - def ok(self) -> bool: - return 200 <= self.status_code < 300 - - -def _auth(key: str) -> Dict[str, str]: - return {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} - - -class ProxyClient: - def __init__( - self, - base_url: str, - master_key: str, - *, - request_timeout: float, - poll_timeout: float, - poll_interval: float, - ) -> None: - self._base_url = base_url.rstrip("/") - self._master_key = master_key - self._request_timeout = request_timeout - self._poll_timeout = poll_timeout - self._poll_interval = poll_interval - - # ---- key / customer management (satisfies lifecycle.ResourceClient) ---- - - def generate_key( - self, - *, - models: Optional[List[str]] = None, - max_budget: Optional[float] = None, - metadata: Optional[Dict[str, object]] = None, - extra_params: Optional[Dict[str, object]] = None, - ) -> str: - payload: Dict[str, object] = {"models": models or [], "duration": None} - if max_budget is not None: - payload["max_budget"] = max_budget - if metadata is not None: - payload["metadata"] = metadata - if extra_params: - payload.update(extra_params) - resp = requests.post( - f"{self._base_url}/key/generate", - headers=_auth(self._master_key), - json=payload, - timeout=self._request_timeout, - ) - resp.raise_for_status() - return str(resp.json()["key"]) - - def key_info(self, key: str) -> Dict[str, object]: - resp = requests.get( - f"{self._base_url}/key/info", - headers=_auth(self._master_key), - params={"key": key}, - timeout=self._request_timeout, - ) - resp.raise_for_status() - return dict(resp.json().get("info", {})) - - def delete_key(self, key: str) -> None: - """Best-effort teardown; a failed cleanup must not fail the test.""" - try: - requests.post( - f"{self._base_url}/key/delete", - headers=_auth(self._master_key), - json={"keys": [key]}, - timeout=self._request_timeout, - ) - except requests.RequestException: - pass - - def delete_customers(self, user_ids: List[str]) -> None: - """Best-effort teardown of end-user/customer rows the `user` param creates.""" - if not user_ids: - return - try: - requests.post( - f"{self._base_url}/customer/delete", - headers=_auth(self._master_key), - json={"user_ids": user_ids}, - timeout=self._request_timeout, - ) - except requests.RequestException: - pass - - # ---- OpenAI-compatible calls ---------------------------------------- - - def chat( - self, - key: str, - model: str, - content: str, - *, - stream: bool = False, - metadata: Optional[Dict[str, object]] = None, - extra_body: Optional[Dict[str, object]] = None, - ) -> CallResult: - body: Dict[str, object] = { - "model": model, - "messages": [{"role": "user", "content": content}], - "stream": stream, - } - if metadata is not None: - body["metadata"] = metadata - if extra_body is not None: - body.update(extra_body) - url = f"{self._base_url}/chat/completions" - if stream: - return self._chat_stream(url, key, body) - resp = requests.post( - url, headers=_auth(key), json=body, timeout=self._request_timeout - ) - parsed = resp.json() if resp.content else {} - choices = parsed.get("choices") or [{}] - message_content = (choices[0].get("message") or {}).get("content") - return CallResult( - status_code=resp.status_code, - call_id=resp.headers.get("x-litellm-call-id"), - response_id=parsed.get("id"), - response_cost_header=resp.headers.get("x-litellm-response-cost"), - body=resp.text, - content=message_content, - ) - - def _chat_stream(self, url: str, key: str, body: Dict[str, object]) -> CallResult: - resp = requests.post( - url, - headers=_auth(key), - json=body, - stream=True, - timeout=self._request_timeout, - ) - if not (200 <= resp.status_code < 300): - return CallResult( - status_code=resp.status_code, - call_id=resp.headers.get("x-litellm-call-id"), - response_id=None, - response_cost_header=resp.headers.get("x-litellm-response-cost"), - body=resp.text, - content=None, - ) - response_id: Optional[str] = None - parts: List[str] = [] - for raw in resp.iter_lines(): - if not raw: - continue - line = raw.decode("utf-8") - if not line.startswith("data:"): - continue - data = line[len("data:") :].strip() - if data == "[DONE]": - break - chunk = json.loads(data) - response_id = chunk.get("id", response_id) - for choice in chunk.get("choices", []): - piece = (choice.get("delta") or {}).get("content") - if piece: - parts.append(piece) - return CallResult( - status_code=resp.status_code, - call_id=resp.headers.get("x-litellm-call-id"), - response_id=response_id, - response_cost_header=resp.headers.get("x-litellm-response-cost"), - body="", - content="".join(parts) or None, - ) - - def embed(self, key: str, model: str, text: str) -> CallResult: - resp = requests.post( - f"{self._base_url}/embeddings", - headers=_auth(key), - json={"model": model, "input": text}, - timeout=self._request_timeout, - ) - parsed = resp.json() if resp.content else {} - return CallResult( - status_code=resp.status_code, - call_id=resp.headers.get("x-litellm-call-id"), - response_id=parsed.get("id"), - response_cost_header=resp.headers.get("x-litellm-response-cost"), - body=resp.text, - content=None, - ) - - # ---- route discovery ------------------------------------------------- - - def get_openapi(self) -> Dict[str, object]: - """The proxy's live route schema from /openapi.json.""" - resp = requests.get( - f"{self._base_url}/openapi.json", timeout=self._request_timeout - ) - resp.raise_for_status() - return dict(resp.json()) - - def probe(self, path: str, params: Optional[Dict[str, str]] = None) -> ProbeResult: - """GET a route with master-key auth; capture status + body to show why.""" - url = f"{self._base_url}{path}" - try: - resp = requests.get( - url, - headers=_auth(self._master_key), - params=params or {}, - timeout=self._request_timeout, - ) - except requests.RequestException as exc: - return ProbeResult(url=url, status_code=-1, body=f"request error: {exc}") - return ProbeResult(url=url, status_code=resp.status_code, body=resp.text) - - # ---- SpendLogs read-back -------------------------------------------- - - def _get_logs( - self, *, request_id: Optional[str] = None, api_key: Optional[str] = None - ) -> List[SpendLogRow]: - params: Dict[str, str] = {} - if request_id is not None: - params["request_id"] = request_id - if api_key is not None: - params["api_key"] = api_key - resp = requests.get( - f"{self._base_url}/spend/logs", - headers=_auth(self._master_key), - params=params, - timeout=self._request_timeout, - ) - if resp.status_code != 200: - return [] - data = resp.json() - return [dict(row) for row in data] if isinstance(data, list) else [] - - def poll_logs_for_key( - self, - key: str, - *, - min_rows: int = 1, - predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None, - ) -> List[SpendLogRow]: - return self._poll(lambda: self._get_logs(api_key=key), min_rows, predicate) - - def poll_logs_for_request_id( - self, - request_id: str, - *, - min_rows: int = 1, - predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None, - ) -> List[SpendLogRow]: - return self._poll( - lambda: self._get_logs(request_id=request_id), min_rows, predicate - ) - - def _poll( - self, - fetch: Callable[[], List[SpendLogRow]], - min_rows: int, - predicate: Optional[Callable[[List[SpendLogRow]], bool]], - ) -> List[SpendLogRow]: - deadline = time.monotonic() + self._poll_timeout - rows: List[SpendLogRow] = [] - while time.monotonic() < deadline: - rows = fetch() - satisfied = len(rows) >= min_rows and ( - predicate is None or predicate(rows) - ) - if satisfied: - return rows - time.sleep(self._poll_interval) - return rows - - -def proxy_client_kwargs() -> Dict[str, object]: - """Constructor kwargs shared by every ProxyClient subclass.""" - return { - "base_url": PROXY_BASE_URL, - "master_key": MASTER_KEY, - "request_timeout": REQUEST_TIMEOUT, - "poll_timeout": POLL_TIMEOUT, - "poll_interval": POLL_INTERVAL, - } - - -def require_successful_call(result: CallOutcome) -> None: - """The skip/fail boundary. - - A non-2xx here means the environment cannot make the call (missing provider - key, upstream down) -> skip. Everything downstream of a 2xx is behavior that - is allowed to fail. - """ - if result.ok: - return - pytest.skip( - f"upstream call unavailable (status {result.status_code}); " - f"body={result.body[:300]}" - ) - - -def unique_marker() -> str: - return uuid.uuid4().hex[:12] diff --git a/tests/e2e_tests/pytest.ini b/tests/e2e_tests/pytest.ini deleted file mode 100644 index 7682dab96ed..00000000000 --- a/tests/e2e_tests/pytest.ini +++ /dev/null @@ -1,7 +0,0 @@ -[pytest] -# Config when any e2e suite under tests/e2e_tests/ is run directly, e.g. -# uv run pytest tests/e2e_tests/spend_tracking/ -v -# The e2e marker is also registered in conftest.py for runs rooted elsewhere. -addopts = --strict-markers --strict-config -markers = - e2e: live test that requires a running proxy and real provider keys diff --git a/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md b/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md deleted file mode 100644 index fbb5ce60c51..00000000000 --- a/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md +++ /dev/null @@ -1,77 +0,0 @@ -# Spend Tracking Test Coverage Matrix - -Scope: every distinct spend-tracking code path, mapped to the test that exercises -it and the level it runs at. Highlights where a live e2e check is the only thing -that would catch a regression. - -Companion: live suite `test_spend_tracking_e2e.py` + route breadth -`test_spend_routes.py` (this directory). Offline regression suite: -`tests/test_litellm/proxy/spend_tracking/`. Reference PR: BerriAI/litellm#29956. - -Levels: `unit` mocked; `integration` real DB/cost-map; `live` real provider + -proxy + SpendLogs rows. Status: `covered` / `partial` / `gap`. - ---- - -## SpendLogs row construction (`spend_tracking_utils.get_logging_payload`) - -| Path | Existing | Level | Status | Live e2e | -|------|----------|-------|--------|----------| -| `_get_status_for_spend_log` | `test_spend_tracking_utils.py` | unit | covered | yes (status read off the row) | -| cache-hit `request_id` suffix | `test_spend_tracking_utils.py` | unit | covered | yes (`test_cache_hit_is_zero_cost_and_suffixed`) | -| failure status + zero spend | `test_spend_tracking_utils.py` | unit | partial | yes (`test_failure_call_writes_failure_status_row`) | -| field population (model/tokens/api_key/team/org) | `test_spend_tracking_utils.py` | unit | partial | yes (asserts real values) | -| `request_tags` propagation | `test_db_spend_update_writer.py` | unit | partial | yes (`test_request_tags_round_trip`) | -| `end_user` attribution | unit | unit | partial | yes (`test_end_user_spend_attributed_on_row`) | - -## Cost calculation by modality - -| Modality | Existing | Status | Live e2e | -|----------|----------|--------|----------| -| Chat (non-stream) | `test_cost_calculator.py`, `local_testing/test_completion_cost.py` | covered | yes (`test_chat_completion_writes_nonzero_spend_row`) | -| Chat (streaming) | `test_streaming_interrupt_spend_tracking.py` | partial | yes (`test_streaming_chat_completion_tracks_spend`) | -| Embedding | `test_cost_calculator.py` (#29956) | partial | yes (`test_embedding_writes_nonzero_spend_row`) | -| Pass-through (gemini/anthropic) | `pass_through_tests/*.test.js` + `llm_translation/` suite | covered | yes (llm_translation suite) | -| Image / audio / rerank / responses / realtime | per-provider unit cost tests | partial/gap | gap | - -## Entity spend aggregation - -| Entity | Existing | Status | Live e2e | -|--------|----------|--------|----------| -| API key | `test_db_spend_update_writer.py`, `test_spend_counters.py` | covered | yes (`test_key_spend_equals_sum_of_logs`) | -| Tag | `test_update_daily_tag_spend.py` | partial | yes (`test_tag_spend_matches_sum_of_tagged_logs`) | -| End-user | `test_proxy_update_spend.py` | covered | yes | -| Spend == sum(logs) consistency | none | gap | yes (key + tag aggregate == sum of rows) | - -## Spend read endpoints (verification surface) - -| Endpoint | Existing | Status | Live e2e | -|----------|----------|--------|----------| -| `/spend/logs` (request_id / api_key) | `test_spend_management_endpoints.py` | covered | yes (primary read path) | -| `/spend/calculate` | `local_testing/test_spend_calculate_endpoint.py` | covered | yes (`test_spend_calculate_returns_nonzero_cost`) | -| `/spend/tags` | `test_spend_management_endpoints.py` | partial | yes (tag accuracy test) | -| whole spend GET surface (22 routes) | unit per-handler | partial | yes (`test_spend_routes.py` probes each for 404/5xx) | - -## What this suite pins - -| Test | Invariant | -|------|-----------| -| `test_chat_completion_writes_nonzero_spend_row` | nonzero cost, token arithmetic, status, row findable by `response.id` | -| `test_streaming_chat_completion_tracks_spend` | streamed responses still costed | -| `test_embedding_writes_nonzero_spend_row` | embedding cost, `completion_tokens == 0` | -| `test_cache_hit_is_zero_cost_and_suffixed` | cache hits not double-charged; `_cache_hit` suffix | -| `test_key_spend_equals_sum_of_logs` | key aggregate == sum of rows | -| `test_request_tags_round_trip` | tags persist onto the row | -| `test_tag_spend_matches_sum_of_tagged_logs` | `/spend/tags` SUM/COUNT == tagged rows | -| `test_end_user_spend_attributed_on_row` | `end_user` attributed + costed | -| `test_failure_call_writes_failure_status_row` | failed call -> `status=failure`, `spend=0` | -| `test_spend_calculate_returns_nonzero_cost` | cost-map smoke (no batch wait) | -| `test_spend_routes.py` (23) | no spend route 404s or 5xxs | - -## Design + timing - -`proxy_batch_write_at` (~60s) means rows land late; every read polls to a deadline. -Fresh scoped key per test (isolation, xdist-safe, cleaned up). Assert invariants -(`spend > 0`, `total == prompt + completion`, aggregate == sum), not literal -$/token values, so pricing drift is not a failure. Skip on environment (no proxy / -no provider key), fail on behavior (a real 2xx call with a wrong/missing row). diff --git a/tests/e2e_tests/spend_tracking/conftest.py b/tests/e2e_tests/spend_tracking/conftest.py deleted file mode 100644 index d0e82691489..00000000000 --- a/tests/e2e_tests/spend_tracking/conftest.py +++ /dev/null @@ -1,16 +0,0 @@ -"""Spend-tracking suite's `client` fixture. - -The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker -live in the parent tests/e2e_tests/conftest.py. SpendE2EClient satisfies the -lifecycle.ResourceClient protocol, so the shared `resources` fixture cleans up -keys and customers this suite creates. -""" - -import pytest - -from spend_e2e_client import SpendE2EClient, build_client - - -@pytest.fixture(scope="session") -def client() -> SpendE2EClient: - return build_client() diff --git a/tests/e2e_tests/spend_tracking/spend_e2e_client.py b/tests/e2e_tests/spend_tracking/spend_e2e_client.py deleted file mode 100644 index f7a1451eb8a..00000000000 --- a/tests/e2e_tests/spend_tracking/spend_e2e_client.py +++ /dev/null @@ -1,94 +0,0 @@ -"""Spend-tracking e2e client: the generic ProxyClient plus spend-specific reads. - -Generic proxy operations (keys, customers, chat/embed, route probing, SpendLogs -polling) live in the shared tests/e2e_tests/proxy_client.py. This module adds only -the spend-specific endpoints: /spend/calculate, /spend/tags, and key-spend polling. - -Re-exports CallResult / SpendLogRow / require_successful_call / unique_marker so -existing tests keep importing them from here. -""" - -import time -from typing import List, Optional - -import requests - -from proxy_client import ( - CallResult, - ProbeResult, - ProxyClient, - SpendLogRow, - _auth, - proxy_client_kwargs, - require_successful_call, - unique_marker, -) - -__all__ = [ - "SpendE2EClient", - "build_client", - "CallResult", - "ProbeResult", - "SpendLogRow", - "require_successful_call", - "unique_marker", -] - - -class SpendE2EClient(ProxyClient): - def calculate_spend(self, model: str, content: str) -> float: - resp = requests.post( - f"{self._base_url}/spend/calculate", - headers=_auth(self._master_key), - json={"model": model, "messages": [{"role": "user", "content": content}]}, - timeout=self._request_timeout, - ) - resp.raise_for_status() - return float(resp.json()["cost"]) - - def spend_by_tags(self) -> List[SpendLogRow]: - """/spend/tags: SUM(spend) GROUP BY tag, straight from SpendLogs.""" - resp = requests.get( - f"{self._base_url}/spend/tags", - headers=_auth(self._master_key), - timeout=self._request_timeout, - ) - if resp.status_code != 200: - return [] - data = resp.json() - if isinstance(data, dict) and "spend_per_tag" in data: - data = data["spend_per_tag"] - return [dict(row) for row in data] if isinstance(data, list) else [] - - def poll_tag_spend( - self, tag: str, *, minimum: float = 0.0 - ) -> Optional[SpendLogRow]: - """Poll until the tag's aggregate spend reaches `minimum`; last seen entry.""" - deadline = time.monotonic() + self._poll_timeout - entry: Optional[SpendLogRow] = None - while time.monotonic() < deadline: - matches = [ - row - for row in self.spend_by_tags() - if str(row.get("individual_request_tag")) == tag - ] - if matches: - entry = matches[0] - if float(entry.get("total_spend") or 0.0) >= minimum: - return entry - time.sleep(self._poll_interval) - return entry - - def poll_key_spend(self, key: str, *, minimum: float = 0.0) -> float: - deadline = time.monotonic() + self._poll_timeout - spend = 0.0 - while time.monotonic() < deadline: - spend = float(self.key_info(key).get("spend") or 0.0) - if spend > minimum: - return spend - time.sleep(self._poll_interval) - return spend - - -def build_client() -> SpendE2EClient: - return SpendE2EClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/spend_tracking/test_spend_routes.py b/tests/e2e_tests/spend_tracking/test_spend_routes.py deleted file mode 100644 index f755eeea0b0..00000000000 --- a/tests/e2e_tests/spend_tracking/test_spend_routes.py +++ /dev/null @@ -1,93 +0,0 @@ -"""Breadth check: query every route on the spend read surface and show what it -returns. - -Spend tracking sprawls across many routes (model-cost / key / user / team / org / -customer aggregation, tags, and activity reports). Most are served with -`include_in_schema=False`, so they do NOT appear in `/openapi.json` - discovery -from the schema alone misses ~70% of the surface. So we probe a curated, verified -list directly, plus any spend route the schema does list (to auto-catch new ones). - -Each probe captures status AND body, so a failure shows the proxy's actual error -(a 500 traceback, a 404 meaning the route was removed) rather than a bare code. -Run with `-rA` (or `-s`) to print every route's response, not just failures. - -Healthy == route exists (not 404) and handler did not crash (not 5xx). A 4xx -(missing params / auth nuance) still means the route is wired and ran. Cheap and -fast: no batch-write wait, no provider calls. -""" - -from datetime import datetime, timedelta, timezone -from typing import Dict, List - -import pytest - -from spend_e2e_client import SpendE2EClient - -pytestmark = pytest.mark.e2e - -# Verified present and responsive on a live proxy. One per row of the spend -# surface: key / user / team / org / customer aggregation, model-cost, tags, -# activity. -SPEND_ROUTES = ( - "/spend/keys", - "/spend/users", - "/spend/tags", - "/spend/logs", - "/spend/logs/ui", - "/global/spend", - "/global/spend/keys", - "/global/spend/teams", - "/global/spend/models", - "/global/spend/provider", - "/global/spend/report", - "/global/spend/tags", - "/global/spend/logs", - "/global/spend/all_tag_names", - "/global/activity", - "/global/activity/model", - "/global/activity/exceptions", - "/key/list", - "/user/list", - "/team/list", - "/organization/list", - "/customer/list", -) - -_SPEND_PREFIXES = ("/spend", "/global/spend", "/global/activity") - - -def _default_params() -> Dict[str, str]: - # Satisfies date-required endpoints (report/activity/provider); ignored elsewhere. - end = datetime.now(timezone.utc).date() - start = end - timedelta(days=1) - return {"start_date": start.isoformat(), "end_date": end.isoformat()} - - -@pytest.mark.parametrize("route", SPEND_ROUTES) -def test_spend_route_responsive(client: SpendE2EClient, route: str) -> None: - result = client.probe(route, params=_default_params()) - print(result) # shown on failure, and for all routes under `-rA` / `-s` - assert result.healthy, str(result) - - -def test_schema_listed_spend_routes_are_responsive(client: SpendE2EClient) -> None: - """Probe any spend GET route the schema lists that isn't in SPEND_ROUTES.""" - paths = client.get_openapi().get("paths", {}) - assert isinstance(paths, dict) and paths, "/openapi.json had no paths" - - discovered: List[str] = [ - path - for path, operations in paths.items() - if isinstance(operations, dict) - and "get" in {m.lower() for m in operations} - and "{" not in path - and any(path.startswith(prefix) for prefix in _SPEND_PREFIXES) - ] - extras = [path for path in discovered if path not in SPEND_ROUTES] - - params = _default_params() - results = [client.probe(path, params=params) for path in extras] - for result in results: - print(result) - offenders = [str(result) for result in results if not result.healthy] - assert not offenders, "non-responsive schema spend routes:\n" + "\n".join(offenders) diff --git a/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py deleted file mode 100644 index 142eb4da40f..00000000000 --- a/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py +++ /dev/null @@ -1,335 +0,0 @@ -"""Live end-to-end spend-tracking tests against a running proxy. - -Run against a proxy started with tests/e2e_tests/docker-config.yaml (or the -gateway config). Coverage rationale: SPEND_TRACKING_COVERAGE_MATRIX.md. - -Model names are literals from that config: chat tests hit "gemini-2.5-flash", -embedding tests hit "openai-text-embedding-3-small". - -Every test: fresh scoped key (isolation) -> real provider call -> -require_successful_call (skip iff env can't make it) -> poll /spend/logs to a -deadline (rows land ~60s later via proxy_batch_write_at) -> assert invariants on -the real row (spend, token arithmetic, status, cache). - -Assertions target invariants, not literals: a regression in the spend pipeline -fails the test; a pricing or token-count drift does not. -""" - -from typing import Callable, Dict, List - -import pytest - -from lifecycle import ResourceManager -from spend_e2e_client import ( - SpendE2EClient, - SpendLogRow, - require_successful_call, - unique_marker, -) - -pytestmark = pytest.mark.e2e - - -def _f(value: object) -> float: - return float(value) if value is not None else 0.0 # type: ignore[arg-type] - - -def _i(value: object) -> int: - return int(value) if value is not None else 0 # type: ignore[arg-type] - - -def _s(value: object) -> str: - return str(value) if value is not None else "" - - -def _summarize(rows: List[SpendLogRow]) -> List[Dict[str, object]]: - keys = ( - "request_id", - "model", - "spend", - "status", - "cache_hit", - "prompt_tokens", - "completion_tokens", - "total_tokens", - ) - return [{k: r.get(k) for k in keys} for r in rows] - - -def _require_row( - rows: List[SpendLogRow], predicate: Callable[[SpendLogRow], bool], what: str -) -> SpendLogRow: - matches = [r for r in rows if predicate(r)] - assert matches, ( - f"no SpendLogs row {what} after polling; saw {len(rows)} row(s): " - f"{_summarize(rows)}" - ) - return matches[0] - - -def test_chat_completion_writes_nonzero_spend_row( - client: SpendE2EClient, scoped_key: str -) -> None: - result = client.chat( - scoped_key, - "gemini-2.5-flash", - f"reply with one word {unique_marker()}", - extra_body={"max_tokens": 16}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(_s(r.get("status")) == "success" for r in rs), - ) - row = _require_row( - rows, lambda r: _s(r.get("status")) == "success", "for the chat call" - ) - - assert _f(row.get("spend")) > 0, f"chat row should cost > 0: {_summarize(rows)}" - assert _s(row.get("status")) == "success" - assert _s(row.get("cache_hit")) != "True", "fresh call must not be a cache hit" - assert "gemini-2.5-flash" in _s(row.get("model")) - - prompt = _i(row.get("prompt_tokens")) - completion = _i(row.get("completion_tokens")) - total = _i(row.get("total_tokens")) - assert prompt > 0 and completion > 0 - assert total == prompt + completion, f"token arithmetic broken: {_summarize(rows)}" - - if result.response_id: - assert any( - _s(r.get("request_id")) == result.response_id for r in rows - ), f"row request_id != client response.id ({result.response_id})" - - -def test_streaming_chat_completion_tracks_spend( - client: SpendE2EClient, scoped_key: str -) -> None: - result = client.chat( - scoped_key, - "gemini-2.5-flash", - f"count to three {unique_marker()}", - stream=True, - extra_body={"max_tokens": 64}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs), - ) - row = _require_row( - rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the stream" - ) - prompt = _i(row.get("prompt_tokens")) - completion = _i(row.get("completion_tokens")) - assert prompt > 0 and completion > 0, f"streaming tokens not tracked: {_summarize(rows)}" - assert _i(row.get("total_tokens")) == prompt + completion - - -def test_embedding_writes_nonzero_spend_row( - client: SpendE2EClient, scoped_key: str -) -> None: - result = client.embed( - scoped_key, - "openai-text-embedding-3-small", - f"vectorize this sentence {unique_marker()}", - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs) - ) - row = _require_row( - rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the embedding" - ) - assert _i(row.get("prompt_tokens")) > 0 - assert _i(row.get("completion_tokens")) == 0, "embeddings have no completion tokens" - assert "text-embedding-3-small" in _s(row.get("model")) - - -def test_cache_hit_is_zero_cost_and_suffixed( - client: SpendE2EClient, scoped_key: str -) -> None: - # Unique marker shared by both calls: call 1 is a guaranteed cache MISS (fresh - # content, paid), call 2 repeats the identical request and HITS the cache just - # populated. The marker keeps each run isolated - a fixed prompt would persist - # in the shared response cache across runs and make both calls hit (flaky). - prompt = f"What is the capital of France? Answer in one word. {unique_marker()}" - first = client.chat( - scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16} - ) - require_successful_call(first) - second = client.chat( - scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16} - ) - require_successful_call(second) - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(_s(r.get("cache_hit")) == "True" for r in rs), - ) - cache_rows = [r for r in rows if _s(r.get("cache_hit")) == "True"] - if not cache_rows: - pytest.skip( - "no cache-hit row observed; caching may be disabled on this proxy. " - f"rows seen: {_summarize(rows)}" - ) - - cache_row = cache_rows[0] - assert _f(cache_row.get("spend")) == 0.0, ( - f"cache hit was charged (double-charge regression): {_summarize(rows)}" - ) - assert "_cache_hit" in _s(cache_row.get("request_id")), ( - "cache-hit row missing the _cache_hit request_id suffix; " - "duplicate-key collisions will silently drop rows" - ) - paid_rows = [r for r in rows if _s(r.get("cache_hit")) != "True"] - assert any(_f(r.get("spend")) > 0 for r in paid_rows), ( - f"the non-cached call should still be charged: {_summarize(rows)}" - ) - - -def test_key_spend_equals_sum_of_logs( - client: SpendE2EClient, scoped_key: str -) -> None: - for _ in range(2): - result = client.chat( - scoped_key, - "gemini-2.5-flash", - f"say hi {unique_marker()}", - extra_body={"max_tokens": 16}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - min_rows=2, - predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0, - ) - assert len(rows) >= 2, f"expected >=2 rows for the key, saw {_summarize(rows)}" - logs_total = sum(_f(r.get("spend")) for r in rows) - assert logs_total > 0 - - key_spend = client.poll_key_spend(scoped_key, minimum=logs_total * 0.999) - assert key_spend == pytest.approx(logs_total, rel=1e-2, abs=1e-9), ( - f"key aggregate {key_spend} != sum of logs {logs_total}; " - f"rows: {_summarize(rows)}" - ) - - -def test_request_tags_round_trip( - client: SpendE2EClient, scoped_key: str -) -> None: - tag = f"e2e-spend-{unique_marker()}" - result = client.chat( - scoped_key, - "gemini-2.5-flash", - "tagged request", - metadata={"tags": [tag]}, - extra_body={"max_tokens": 16}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(tag in _s(r.get("request_tags")) for r in rs), - ) - _require_row( - rows, - lambda r: tag in _s(r.get("request_tags")), - f"carrying request tag {tag!r}", - ) - - -def test_tag_spend_matches_sum_of_tagged_logs( - client: SpendE2EClient, scoped_key: str -) -> None: - # Unique tag so /spend/tags can't be polluted by other rows; unique content - # per call so both are fresh misses (paid), not cache hits. - tag = f"e2e-tagspend-{unique_marker()}" - for _ in range(2): - result = client.chat( - scoped_key, - "gemini-2.5-flash", - f"hi {unique_marker()}", - metadata={"tags": [tag]}, - extra_body={"max_tokens": 16}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - min_rows=2, - predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0, - ) - tagged = [r for r in rows if tag in _s(r.get("request_tags"))] - assert len(tagged) >= 2, f"expected 2 tagged rows, saw {_summarize(rows)}" - logs_total = sum(_f(r.get("spend")) for r in tagged) - assert logs_total > 0 - - entry = client.poll_tag_spend(tag, minimum=logs_total * 0.999) - assert entry is not None, f"tag {tag!r} never appeared in /spend/tags" - assert _f(entry.get("total_spend")) == pytest.approx( - logs_total, rel=1e-2, abs=1e-9 - ), f"/spend/tags total_spend {entry} != sum of tagged rows {logs_total}" - assert _i(entry.get("log_count")) == len(tagged), ( - f"/spend/tags log_count {entry.get('log_count')} != tagged rows {len(tagged)}" - ) - - -def test_end_user_spend_attributed_on_row( - client: SpendE2EClient, scoped_key: str, resources: ResourceManager -) -> None: - customer = resources.customer(f"e2e-cust-{unique_marker()}") - result = client.chat( - scoped_key, - "gemini-2.5-flash", - "hi", - extra_body={"user": customer, "max_tokens": 16}, - ) - require_successful_call(result) - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(_s(r.get("end_user")) == customer for r in rs), - ) - row = _require_row( - rows, - lambda r: _s(r.get("end_user")) == customer, - f"attributed to end_user {customer!r}", - ) - assert _f(row.get("spend")) > 0, f"end-user row should cost > 0: {_summarize(rows)}" - - -def test_failure_call_writes_failure_status_row( - client: SpendE2EClient, scoped_key: str -) -> None: - result = client.chat( - scoped_key, "gemini-2.5-flash", "", extra_body={"max_tokens": 1} - ) - if result.ok: - pytest.skip("call unexpectedly succeeded; could not induce a failure row") - - rows = client.poll_logs_for_key( - scoped_key, - predicate=lambda rs: any(_s(r.get("status")) == "failure" for r in rs), - ) - failure_rows = [r for r in rows if _s(r.get("status")) == "failure"] - if not failure_rows: - pytest.skip( - "no failure-status row was logged for the rejected call; " - "failure logging is environment-specific" - ) - assert _f(failure_rows[0].get("spend")) == 0.0, "failed call must not be charged" - - -def test_spend_calculate_returns_nonzero_cost(client: SpendE2EClient) -> None: - cost = client.calculate_spend( - "gemini-2.5-flash", "estimate the cost of this request" - ) - assert cost > 0, ( - "/spend/calculate returned 0 for gemini-2.5-flash; " - "cost map may be missing this model" - )