From 2e4066be4d2ae0c705d3ea18039331e5b41a1d4e Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Thu, 18 Jun 2026 15:08:24 -0700 Subject: [PATCH] tests: add e2e tests for spend, budgets and llms --- tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md | 93 +++++ .../budgets/BUDGET_TEST_COVERAGE_MATRIX.md | 78 ++++ tests/e2e_tests/budgets/budget_client.py | 185 +++++++++ tests/e2e_tests/budgets/conftest.py | 16 + .../e2e_tests/budgets/test_budget_crud_e2e.py | 68 ++++ .../budgets/test_budget_enforcement_e2e.py | 119 ++++++ .../budgets/test_model_max_budget_e2e.py | 60 +++ .../e2e_tests/budgets/test_soft_budget_e2e.py | 36 ++ .../e2e_tests/budgets/test_tag_budget_e2e.py | 58 +++ tests/e2e_tests/conftest.py | 54 +++ tests/e2e_tests/docker-compose.yml | 59 +++ tests/e2e_tests/docker-config.yaml | 39 ++ tests/e2e_tests/e2e_config.py | 16 + tests/e2e_tests/lifecycle.py | 87 ++++ .../LLM_TRANSLATION_COVERAGE_MATRIX.md | 84 ++++ tests/e2e_tests/llm_translation/conftest.py | 16 + .../llm_translation/passthrough_client.py | 136 +++++++ .../llm_translation/test_passthrough_e2e.py | 158 ++++++++ tests/e2e_tests/proxy_client.py | 371 ++++++++++++++++++ tests/e2e_tests/pytest.ini | 7 + .../SPEND_TRACKING_COVERAGE_MATRIX.md | 77 ++++ tests/e2e_tests/spend_tracking/conftest.py | 16 + .../spend_tracking/spend_e2e_client.py | 94 +++++ .../spend_tracking/test_spend_routes.py | 93 +++++ .../spend_tracking/test_spend_tracking_e2e.py | 335 ++++++++++++++++ 25 files changed, 2355 insertions(+) create mode 100644 tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md create mode 100644 tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md create mode 100644 tests/e2e_tests/budgets/budget_client.py create mode 100644 tests/e2e_tests/budgets/conftest.py create mode 100644 tests/e2e_tests/budgets/test_budget_crud_e2e.py create mode 100644 tests/e2e_tests/budgets/test_budget_enforcement_e2e.py create mode 100644 tests/e2e_tests/budgets/test_model_max_budget_e2e.py create mode 100644 tests/e2e_tests/budgets/test_soft_budget_e2e.py create mode 100644 tests/e2e_tests/budgets/test_tag_budget_e2e.py create mode 100644 tests/e2e_tests/conftest.py create mode 100644 tests/e2e_tests/docker-compose.yml create mode 100644 tests/e2e_tests/docker-config.yaml create mode 100644 tests/e2e_tests/e2e_config.py create mode 100644 tests/e2e_tests/lifecycle.py create mode 100644 tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md create mode 100644 tests/e2e_tests/llm_translation/conftest.py create mode 100644 tests/e2e_tests/llm_translation/passthrough_client.py create mode 100644 tests/e2e_tests/llm_translation/test_passthrough_e2e.py create mode 100644 tests/e2e_tests/proxy_client.py create mode 100644 tests/e2e_tests/pytest.ini create mode 100644 tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md create mode 100644 tests/e2e_tests/spend_tracking/conftest.py create mode 100644 tests/e2e_tests/spend_tracking/spend_e2e_client.py create mode 100644 tests/e2e_tests/spend_tracking/test_spend_routes.py create mode 100644 tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py diff --git a/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md b/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md new file mode 100644 index 00000000000..63cb41228d2 --- /dev/null +++ b/tests/e2e_tests/budgets/BUDGET_CODE_MATRIX.md @@ -0,0 +1,93 @@ +# Budget Code Matrix + +What LiteLLM actually implements for budgets: every entity that can carry a dollar +budget, how the limit is enforced, and where in the code it happens. This is the +"what we support" reference; the companion `BUDGET_TEST_COVERAGE_MATRIX.md` maps +each row to its tests and the e2e gaps. + +Over-budget surfaces as a `budget_exceeded` error (the live suite +`tests/otel_tests/test_e2e_budgeting.py` asserts `type == "budget_exceeded"`, +`code == "429"`); the underlying `BudgetExceededError` is defined in +`litellm/exceptions.py` (`status_code=400`). Enforcement runs in `common_checks()` +/ `auth_checks.py` at auth time, plus pre-call reservation in +`budget_reservation.py`. + +Legend for "Enforced": **block** = request rejected; **filter** = router skips the +deployment; **alert** = notify only, request proceeds. + +--- + +## 1. Per-entity dollar budgets + +| Entity | Budget stored | Hard `max_budget` | Soft budget | Per-window | Model budget | Reset by `budget_duration` | +|--------|---------------|-------------------|-------------|------------|--------------|----------------------------| +| API key | `LiteLLM_VerificationToken` (direct cols + `budget_id` FK) | block (`_virtual_key_max_budget_check`) | alert (`_virtual_key_soft_budget_check`) + 80% alert | block (`_virtual_key_multi_budget_check`) | block (`model_max_budget_limiter.is_key_within_model_budget`) | keys reset job | +| Internal user | `LiteLLM_UserTable` (direct cols) | block (`common_checks`, only when not on a team) | - | - | via `model_max_budget` json | users reset job | +| Team | `LiteLLM_TeamTable` (direct cols) | block (`_team_max_budget_check`) | alert (`_team_soft_budget_check`) | block (`_team_multi_budget_check`) | via `model_max_budget` | teams reset job | +| Team member | `LiteLLM_TeamMembership` -> `LiteLLM_BudgetTable` | block (`_check_team_member_budget`) | - | - | - | budget-table reset job | +| End-user / customer | `LiteLLM_EndUserTable` -> `LiteLLM_BudgetTable` | block (`_check_end_user_budget`) | - | - | block (`is_end_user_within_model_budget`) | budget-table reset job | +| Organization | `LiteLLM_OrganizationTable` -> `LiteLLM_BudgetTable` | block (`_organization_max_budget_check`) | - | - | via budget-table | budget-table reset job | +| Tag | `LiteLLM_TagTable` -> `LiteLLM_BudgetTable` | block (`_tag_max_budget_check`) | - | - | via budget-table | budget-table reset job | +| Project | `LiteLLM_ProjectTable` -> `LiteLLM_BudgetTable` | block (`_project_max_budget_check`) | alert (`_project_soft_budget_check`) | - | - | budget-table reset job | +| Provider (router) | config `provider_budget_config` (in-memory) | filter (`router_strategy/budget_limiter`) | - | yes (time window) | - | window TTL | +| Global proxy | `litellm.max_budget` (config) | block (`_global_proxy_budget_check`) | - | - | - | - | + +Notes / flags from the code: +- **User budget only enforced off-team**: `common_checks` skips the personal-user + budget when the key belongs to a team (team budget governs instead). +- **Comparison operators are inconsistent**: key/user use `>=`, team/end-user main + budget use `>`. Spend exactly at `max_budget` blocks a key but not a team. +- **Provider budgets are filter-only**: an over-budget provider is removed from + routing; if all are over budget the router raises + `no_deployments_with_provider_budget_routing` (not a per-entity block). +- **Enforcement timing differs by entity**: key / user / org / team-member / tag / + model enforce off real-time reservation counters (block within ~2 calls); + **end-user** enforcement reads `EndUserTable.spend`, which only updates on the + `proxy_batch_write_at` flush, so it lags by that interval (verified live). + +## 2. Budget mechanisms + +| Mechanism | What it does | Code | +|-----------|--------------|------| +| Pre-call reservation | Estimates max request cost, atomically reserves against redis spend counters for key/team/user/end_user/tag/team_member/org before the call; blocks if a counter would exceed | `spend_tracking/budget_reservation.py` | +| Post-call reconciliation | Adjusts the reservation to the actual cost once known | `reconcile_budget_reservation` | +| Read-time enforcement | Auth-time check of current spend vs `max_budget` | `auth_checks.common_checks` + per-entity `_*_max_budget_check` | +| Soft budget / alerts | At `soft_budget` (or 80% of max) fire Slack/email alert, do not block | `_virtual_key_soft_budget_check`, `_team_soft_budget_check`, `budget_alerts` | +| Multi-window budgets | `budget_limits` list of `{budget_duration, max_budget}`; each window enforced + reset independently | `_virtual_key_multi_budget_check`, `reset_budget_windows` | +| Model-level budgets | `model_max_budget` dict (per model: `budget_limit` + `time_period`) on key/user/end_user | `hooks/model_max_budget_limiter.py` | +| Reset by duration | Job zeros `spend`, recomputes `budget_reset_at = now + duration_in_seconds(budget_duration)`, invalidates redis counters | `common_utils/reset_budget_job.py`, `duration_parser.duration_in_seconds` | +| Zero-cost bypass | Models with no configured price bypass budget reservation | `budget_reservation` zero-cost path | + +## 3. Budget management surface (endpoints) + +| Action | Endpoint | Handler | +|--------|----------|---------| +| Create budget | `POST /budget/new` | `new_budget` | +| Update budget | `POST /budget/update` | `update_budget` | +| Budget info | `POST /budget/info` (`{"budgets": [id]}`) | `info_budget` | +| Budget settings | `GET /budget/settings` | `budget_settings` | +| List budgets | `GET /budget/list` | `list_budget` | +| Delete budget | `POST /budget/delete` (`{"id": id}`) | `delete_budget` | +| Set on key | `POST /key/generate`, `/key/update` (`max_budget`, `soft_budget`, `budget_duration`, `model_max_budget`, `budget_id`) | key mgmt | +| Set on user | `POST /user/new` (`max_budget`, `budget_duration`) | internal user | +| Set on team | `POST /team/new` (`max_budget`, `soft_budget`, `team_member_budget`) | team | +| Set on team member | `POST /team/member_add` (`max_budget_in_team`) | team | +| Set on org | `POST /organization/new` (`max_budget`, `soft_budget`, `model_max_budget`) | org | +| Set on customer | `POST /customer/new`, `/customer/update` (`max_budget`, `budget_id`) | customer | +| Set on tag | `POST /tag/new`, `/tag/update` (`max_budget`) | tag mgmt | +| Read budget+spend | `/key/info`, `/user/info`, `/team/info`, `/organization/info`, `/customer/info`, `/budget/info` | per-entity info | + +Endpoint method/shape gotchas verified live: `/organization/delete` is **DELETE** +with `{"organization_ids": [id]}`; `/budget/info` takes `{"budgets": [id]}`; +`model_max_budget` entries use `{"budget_limit", "time_period"}`. + +## 4. Config knobs + +| Setting | Effect | +|---------|--------| +| `litellm.max_budget` | proxy-wide hard cap (global proxy budget) | +| `max_internal_user_budget` / `default_max_internal_user_budget` | default `max_budget` for internal users | +| `internal_user_budget_duration` | default reset duration for internal users | +| `max_end_user_budget` / `max_end_user_budget_id` | default budget for end-users | +| `default_team_params` | default `max_budget` / `budget_duration` / limits for teams | +| `provider_budget_config` (router) | per-provider spend caps + windows | diff --git a/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md b/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md new file mode 100644 index 00000000000..0704ee9b919 --- /dev/null +++ b/tests/e2e_tests/budgets/BUDGET_TEST_COVERAGE_MATRIX.md @@ -0,0 +1,78 @@ +# Budget Test Coverage Matrix + +Maps every row of `BUDGET_CODE_MATRIX.md` (what LiteLLM implements) to its tests +and level, then marks the live e2e coverage this suite adds. + +Levels: `unit` mocked (`AsyncMock` on `get_current_spend`/prisma); `router` live +router with fake deployments; `live-e2e` real proxy, real key/team, real requests +until blocked. Status: `covered` / `partial` / `gap`. + +Pre-existing live coverage outside this suite: +- `tests/otel_tests/test_e2e_budgeting.py` - key + team enforcement, budget update. +- `tests/local_testing/test_router_budget_limiter.py` - provider / tag / deployment + budgets at the router. + +This suite (`tests/e2e_tests/budgets/`) adds the missing live coverage and runs +on the shared lifecycle (every entity it creates is deleted on teardown). + +--- + +## Per-entity enforcement + +| Entity | Unit | Pre-existing live | This suite (live) | Status | +|--------|------|-------------------|-------------------|--------| +| API key | `test_budget_reservation.py`, `test_max_budget_limiter.py` | `otel_tests` | `test_budget_enforcement_e2e::test_key_budget_blocks` | **covered** | +| Team | `test_team_budget_limits.py` | `otel_tests` | (org test builds a team) | **covered** | +| Internal user | auth unit tests | - | `test_internal_user_budget_blocks` | **covered (new)** | +| Team member | `test_team_member_budget.py` | - | `test_team_member_budget_blocks` | **covered (new)** | +| End-user / customer | `test_custom_auth_end_user_budget.py` | - | `test_end_user_budget_blocks` | **covered (new)** | +| Organization | `test_organization_budget_enforcement.py` (flagged weak) | - | `test_organization_budget_blocks` | **covered (new)** | +| Tag (proxy-level) | - | router only | `test_tag_budget_e2e::test_tag_budget_blocks_tagged_requests` | **covered (new)** | +| Model-level (`model_max_budget`) | `test_unit_test_max_model_budget_limiter.py` | - | `test_model_max_budget_e2e::test_model_max_budget_isolates_per_model` | **covered (new)** | +| Provider (router) | `test_budget_limiter_hotpath.py` | `test_router_budget_limiter.py` | - | **covered** (router) | +| Global proxy (`litellm.max_budget`) | unit | - | - | **gap** (needs a config-level cap; not key-settable) | + +## Budget mechanisms + +| Mechanism | Unit | This suite (live) | Status | +|-----------|------|-------------------|--------| +| Pre-call reservation | `test_budget_reservation.py` | exercised by every enforcement test | **partial** | +| Soft budget / alerts | `SlackAlerting/test_budget_alert_types.py` | `test_soft_budget_e2e::test_soft_budget_does_not_block` | **covered (new)** (block-vs-alert; the alert side-effect itself stays unit) | +| Budget CRUD | `test_budget_endpoints.py` | `test_budget_crud_e2e` (roundtrip + delete) | **covered (new)** | +| Reset scheduling | `test_proxy_budget_reset.py` | `test_budget_crud_e2e::test_budget_duration_schedules_reset_on_key` | **covered (new)** (scheduling; actual zeroing is time-dependent -> unit) | +| Multi-window budgets | `test_multi_budget_windows.py` | - | **gap** (window setup is fiddly; left to unit for now) | +| Read budget+spend | `test_spend_management_endpoints.py` | `/key/info` asserted in CRUD + enforcement | **partial** | + +## Remaining gaps (intentionally not live-tested) + +- **Global proxy budget** (`litellm.max_budget`): set via proxy config, not a + per-key API, so it needs a dedicated proxy boot with that config rather than a + runtime-created entity. Out of scope for the per-entity suite. +- **Multi-window budgets**: the `budget_limits` list shape and per-window reset are + covered by `test_multi_budget_windows.py` (unit); a live version would need to + wait out a short window to see the reset, which is time-dependent. +- **Soft-budget alert delivery**: whether the Slack/email actually fires is not + observable from the proxy API; unit tests own that. The live test pins the + load-bearing behavior (soft does not block). +- **Reset zeroing after the window elapses**: time-dependent; unit tests own the + reset-job logic. The live test pins that `budget_reset_at` is scheduled. + +## This suite's files + +| File | Covers | +|------|--------| +| `test_budget_enforcement_e2e.py` | key / internal-user / end-user / organization / team-member hard enforcement | +| `test_model_max_budget_e2e.py` | per-model caps isolate by model | +| `test_soft_budget_e2e.py` | soft budget alerts but does not block | +| `test_tag_budget_e2e.py` | proxy-level tag budget blocks tagged requests, spares others | +| `test_budget_crud_e2e.py` | `/budget/*` CRUD roundtrip + delete + `budget_reset_at` scheduling | + +## Pattern + timing + +Create the entity with a tiny `max_budget`, drive spend until a `budget_exceeded` +block. The enforcement helper is two-phase: a fast warmup (key/user/org/member/tag/ +model block within ~2 calls off real-time counters), then a poll across the ~60s +batch-write window (end-user enforcement reads table spend that lags). Skip on a +non-budget error (provider down / key missing); fail if the budget is never +enforced. Chat tests use `gpt-5.5` (the model with a working key on the reference +proxy); swap the literal if your proxy differs. diff --git a/tests/e2e_tests/budgets/budget_client.py b/tests/e2e_tests/budgets/budget_client.py new file mode 100644 index 00000000000..2ca65462d2f --- /dev/null +++ b/tests/e2e_tests/budgets/budget_client.py @@ -0,0 +1,185 @@ +"""Client for budget e2e tests: the shared ProxyClient plus budget-bearing entity +management (user / team / team-member / org / customer / tag / budget-table) and +info reads. + +Over-budget surfaces as a ``budget_exceeded`` error; ``is_budget_block`` detects it +on a CallResult. Create methods return the new id and raise on failure; tests +register the matching delete with ``resources.defer(...)`` for cleanup. +""" + +from typing import Dict, List, Optional + +import requests + +from proxy_client import CallResult, ProxyClient, _auth, proxy_client_kwargs + + +def is_budget_block(result: CallResult) -> bool: + """True if the call was rejected for being over budget (vs a provider error).""" + return not result.ok and "budget_exceeded" in result.body + + +def model_budget(model: str, limit: float, period: str = "30d") -> dict: + """A model_max_budget dict entry: per-model cap with a reset window.""" + return {model: {"budget_limit": limit, "time_period": period}} + + +class BudgetClient(ProxyClient): + def _post(self, path: str, body: Dict[str, object]) -> Dict[str, object]: + resp = requests.post( + f"{self._base_url}{path}", + headers=_auth(self._master_key), + json=body, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return dict(resp.json()) if resp.text else {} + + def _get(self, path: str, params: Dict[str, str]) -> Dict[str, object]: + resp = requests.get( + f"{self._base_url}{path}", + headers=_auth(self._master_key), + params=params, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return dict(resp.json()) + + def _delete(self, path: str, body: Dict[str, object]) -> None: + try: + requests.delete( + f"{self._base_url}{path}", + headers=_auth(self._master_key), + json=body, + timeout=self._request_timeout, + ) + except requests.RequestException: + pass + + def _post_quiet(self, path: str, body: Dict[str, object]) -> None: + try: + requests.post( + f"{self._base_url}{path}", + headers=_auth(self._master_key), + json=body, + timeout=self._request_timeout, + ) + except requests.RequestException: + pass + + # ---- internal user -------------------------------------------------- + + def create_user( + self, *, max_budget: float, budget_duration: Optional[str] = None + ) -> str: + body: Dict[str, object] = {"max_budget": max_budget} + if budget_duration is not None: + body["budget_duration"] = budget_duration + return str(self._post("/user/new", body)["user_id"]) + + def delete_user(self, user_id: str) -> None: + self._post_quiet("/user/delete", {"user_ids": [user_id]}) + + def user_info(self, user_id: str) -> Dict[str, object]: + return self._get("/user/info", {"user_id": user_id}) + + # ---- customer / end-user ------------------------------------------- + + def create_customer(self, customer_id: str, *, max_budget: float) -> str: + self._post("/customer/new", {"user_id": customer_id, "max_budget": max_budget}) + return customer_id + + def customer_info(self, customer_id: str) -> Dict[str, object]: + return self._get("/customer/info", {"end_user_id": customer_id}) + + # ---- organization --------------------------------------------------- + + def create_org(self, *, max_budget: float, alias: str) -> str: + return str( + self._post( + "/organization/new", + {"organization_alias": alias, "max_budget": max_budget}, + )["organization_id"] + ) + + def delete_org(self, org_id: str) -> None: + self._delete("/organization/delete", {"organization_ids": [org_id]}) + + # ---- team ----------------------------------------------------------- + + def create_team( + self, + *, + alias: str, + max_budget: Optional[float] = None, + organization_id: Optional[str] = None, + extra: Optional[Dict[str, object]] = None, + ) -> str: + body: Dict[str, object] = {"team_alias": alias} + if max_budget is not None: + body["max_budget"] = max_budget + if organization_id is not None: + body["organization_id"] = organization_id + if extra: + body.update(extra) + return str(self._post("/team/new", body)["team_id"]) + + def delete_team(self, team_id: str) -> None: + self._post_quiet("/team/delete", {"team_ids": [team_id]}) + + def team_info(self, team_id: str) -> Dict[str, object]: + return self._get("/team/info", {"team_id": team_id}) + + def add_team_member( + self, team_id: str, user_id: str, *, max_budget_in_team: Optional[float] = None + ) -> None: + body: Dict[str, object] = { + "team_id": team_id, + "member": {"role": "user", "user_id": user_id}, + } + if max_budget_in_team is not None: + body["max_budget_in_team"] = max_budget_in_team + self._post("/team/member_add", body) + + # ---- tag ------------------------------------------------------------ + + def create_tag(self, name: str, *, max_budget: float) -> str: + self._post("/tag/new", {"name": name, "max_budget": max_budget}) + return name + + def delete_tag(self, name: str) -> None: + self._post_quiet("/tag/delete", {"name": name}) + + # ---- budget table --------------------------------------------------- + + def create_budget( + self, + *, + max_budget: float, + soft_budget: Optional[float] = None, + budget_duration: Optional[str] = None, + ) -> str: + body: Dict[str, object] = {"max_budget": max_budget} + if soft_budget is not None: + body["soft_budget"] = soft_budget + if budget_duration is not None: + body["budget_duration"] = budget_duration + return str(self._post("/budget/new", body)["budget_id"]) + + def delete_budget(self, budget_id: str) -> None: + self._post_quiet("/budget/delete", {"id": budget_id}) + + def budget_info(self, budget_id: str) -> List[Dict[str, object]]: + resp = requests.post( + f"{self._base_url}/budget/info", + headers=_auth(self._master_key), + json={"budgets": [budget_id]}, + timeout=self._request_timeout, + ) + resp.raise_for_status() + data = resp.json() + return [dict(row) for row in data] if isinstance(data, list) else [] + + +def build_client() -> BudgetClient: + return BudgetClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/budgets/conftest.py b/tests/e2e_tests/budgets/conftest.py new file mode 100644 index 00000000000..12421db91e9 --- /dev/null +++ b/tests/e2e_tests/budgets/conftest.py @@ -0,0 +1,16 @@ +"""Budgets suite's `client` fixture. + +The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker +live in the parent tests/e2e_tests/conftest.py. BudgetClient subclasses +ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources` +fixture cleans up keys; tests register entity deletes via `resources.defer(...)`. +""" + +import pytest + +from budget_client import BudgetClient, build_client + + +@pytest.fixture(scope="session") +def client() -> BudgetClient: + return build_client() diff --git a/tests/e2e_tests/budgets/test_budget_crud_e2e.py b/tests/e2e_tests/budgets/test_budget_crud_e2e.py new file mode 100644 index 00000000000..459d019b18e --- /dev/null +++ b/tests/e2e_tests/budgets/test_budget_crud_e2e.py @@ -0,0 +1,68 @@ +"""Live e2e for the budget management surface (no LLM calls, fast). + +Covers the budget-table CRUD round-trip and that `budget_duration` schedules a +`budget_reset_at`. The actual zeroing after the window is time-dependent, so we +assert the reset is *scheduled* (now + duration), not waited out. +""" + +from datetime import datetime, timezone + +import pytest + +from budget_client import BudgetClient +from lifecycle import ResourceManager + +pytestmark = pytest.mark.e2e + + +def _f(value: object) -> float: + return float(value) if value is not None else 0.0 # type: ignore[arg-type] + + +def test_budget_crud_roundtrip( + client: BudgetClient, resources: ResourceManager +) -> None: + budget_id = client.create_budget( + max_budget=12.5, soft_budget=10.0, budget_duration="30d" + ) + resources.defer(lambda: client.delete_budget(budget_id)) + + rows = client.budget_info(budget_id) + assert rows, f"/budget/info returned nothing for {budget_id}" + row = rows[0] + assert _f(row.get("max_budget")) == 12.5 + assert _f(row.get("soft_budget")) == 10.0 + assert row.get("budget_reset_at"), "budget_duration did not schedule a reset" + + # Attach the budget to a key and confirm the key reflects it. + key = client.generate_key(extra_params={"budget_id": budget_id}) + resources.defer(lambda: client.delete_key(key)) + info = client.key_info(key) + linked = info.get("litellm_budget_table") or {} + assert ( + info.get("budget_id") == budget_id or _f(linked.get("max_budget")) == 12.5 + ), f"key does not reflect attached budget: {info.get('budget_id')}, {linked}" + + +def test_budget_delete_removes_it( + client: BudgetClient, resources: ResourceManager +) -> None: + budget_id = client.create_budget(max_budget=1.0) + client.delete_budget(budget_id) + assert client.budget_info(budget_id) == [], "budget still present after delete" + + +def test_budget_duration_schedules_reset_on_key( + client: BudgetClient, resources: ResourceManager +) -> None: + key = client.generate_key( + max_budget=10.0, extra_params={"budget_duration": "30d"} + ) + resources.defer(lambda: client.delete_key(key)) + + reset_at = client.key_info(key).get("budget_reset_at") + assert reset_at, "budget_duration did not set budget_reset_at on the key" + + reset_dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00")) + days_out = (reset_dt - datetime.now(timezone.utc)).total_seconds() / 86400 + assert 28 < days_out < 32, f"reset ~30d expected, got {days_out:.1f}d out" diff --git a/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py b/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py new file mode 100644 index 00000000000..591776ca22e --- /dev/null +++ b/tests/e2e_tests/budgets/test_budget_enforcement_e2e.py @@ -0,0 +1,119 @@ +"""Live e2e: a tiny max_budget on an entity actually blocks requests. + +Mirrors tests/otel_tests/test_e2e_budgeting.py (call until `budget_exceeded`) but +covers the entities that had NO live coverage - internal user, end-user, +organization, team member - and runs in this suite so the shared `resources` +teardown deletes every entity created. See BUDGET_TEST_COVERAGE_MATRIX.md. + +Skip on environment, fail on behavior: a non-budget error (provider down) skips; +if calls never get blocked, budget enforcement is broken -> fail. +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +TINY_BUDGET = 1e-6 # one real call's spend exceeds this, so the block lands fast + + +def _assert_budget_blocks( + client: BudgetClient, key: str, model: str, *, user: str = "" +) -> None: + """Drive spend over the entity's budget and assert a call is blocked. + + Two-phase so it's robust across entities: key/user/org enforce off real-time + reservation counters (block within a couple calls), but end-user enforcement + reads the table spend that only updates on the ~60s batch write - so after a + warmup we poll for the block across that window. Skip on a non-budget error; + fail if the budget is never enforced. + """ + extra: dict = {"max_tokens": 16} + if user: + extra["user"] = user + + def _call(label: str): + result = client.chat(key, model, f"{label} {unique_marker()}", extra_body=extra) + if not result.ok and not is_budget_block(result): + require_successful_call(result) # non-budget error -> skip + return result + + for _ in range(2): # warmup: incur spend (fast entities block here) + if is_budget_block(_call("warmup")): + return + time.sleep(0.5) + + deadline = time.monotonic() + 100 # cover the batch-write propagation window + while time.monotonic() < deadline: + if is_budget_block(_call("probe")): + return + time.sleep(5) + pytest.fail("budget not enforced within 100s") + + +def test_key_budget_blocks(client: BudgetClient, resources: ResourceManager) -> None: + key = client.generate_key(max_budget=TINY_BUDGET) + resources.defer(lambda: client.delete_key(key)) + _assert_budget_blocks(client, key, "gpt-5.5") + + +def test_internal_user_budget_blocks( + client: BudgetClient, resources: ResourceManager +) -> None: + user_id = client.create_user(max_budget=TINY_BUDGET) + resources.defer(lambda: client.delete_user(user_id)) + # personal key (no team) -> the user budget governs + key = client.generate_key(extra_params={"user_id": user_id}) + resources.defer(lambda: client.delete_key(key)) + _assert_budget_blocks(client, key, "gpt-5.5") + + +def test_end_user_budget_blocks( + client: BudgetClient, scoped_key: str, resources: ResourceManager +) -> None: + customer = f"e2e-budget-cust-{unique_marker()}" + client.create_customer(customer, max_budget=TINY_BUDGET) + resources.defer(lambda: client.delete_customers([customer])) + _assert_budget_blocks(client, scoped_key, "gpt-5.5", user=customer) + + +def test_organization_budget_blocks( + client: BudgetClient, resources: ResourceManager +) -> None: + # Org carries the tiny budget; the team under it has none, so a block here is + # org-level enforcement (the historically weak link). + org_id = client.create_org( + max_budget=TINY_BUDGET, alias=f"e2e-budget-org-{unique_marker()}" + ) + resources.defer(lambda: client.delete_org(org_id)) + team_id = client.create_team( + alias=f"e2e-budget-team-{unique_marker()}", organization_id=org_id + ) + resources.defer(lambda: client.delete_team(team_id)) + key = client.generate_key(extra_params={"team_id": team_id}) + resources.defer(lambda: client.delete_key(key)) + _assert_budget_blocks(client, key, "gpt-5.5") + + +def test_team_member_budget_blocks( + client: BudgetClient, resources: ResourceManager +) -> None: + # Member's per-team budget is tiny while the team itself has a large budget, + # so a block proves member-level (not team-level) enforcement. + team_id = client.create_team( + alias=f"e2e-budget-team-{unique_marker()}", max_budget=100.0 + ) + resources.defer(lambda: client.delete_team(team_id)) + user_id = client.create_user(max_budget=100.0) + resources.defer(lambda: client.delete_user(user_id)) + client.add_team_member(team_id, user_id, max_budget_in_team=TINY_BUDGET) + key = client.generate_key( + extra_params={"team_id": team_id, "user_id": user_id} + ) + resources.defer(lambda: client.delete_key(key)) + _assert_budget_blocks(client, key, "gpt-5.5") diff --git a/tests/e2e_tests/budgets/test_model_max_budget_e2e.py b/tests/e2e_tests/budgets/test_model_max_budget_e2e.py new file mode 100644 index 00000000000..7db8c3fec2e --- /dev/null +++ b/tests/e2e_tests/budgets/test_model_max_budget_e2e.py @@ -0,0 +1,60 @@ +"""Live e2e: per-model budgets (`model_max_budget`) isolate by model. + +A key caps one model tiny and leaves another generous. Exhausting the capped +model must block *that* model while the other still works - proving the per-model +cap is enforced independently, not as a key-wide budget. Closes the +model_max_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block, model_budget +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +CAPPED_MODEL = "gpt-5.5" +FREE_MODEL = "claude-haiku-4-5" + + +def _call(client: BudgetClient, key: str, model: str): + result = client.chat( + key, model, f"hi {unique_marker()}", extra_body={"max_tokens": 16} + ) + if not result.ok and not is_budget_block(result): + require_successful_call(result) # non-budget error -> skip + return result + + +def test_model_max_budget_isolates_per_model( + client: BudgetClient, resources: ResourceManager +) -> None: + key = client.generate_key( + extra_params={ + "model_max_budget": { + **model_budget(CAPPED_MODEL, 1e-6), + **model_budget(FREE_MODEL, 1000.0), + } + } + ) + resources.defer(lambda: client.delete_key(key)) + + # Exhaust the capped model. + blocked = False + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + if is_budget_block(_call(client, key, CAPPED_MODEL)): + blocked = True + break + time.sleep(1) + assert blocked, f"{CAPPED_MODEL} per-model budget never enforced" + + # The other model shares the key but has its own (large) cap -> still works. + other = _call(client, key, FREE_MODEL) + require_successful_call(other) # skips if claude key absent; never a budget block + assert not is_budget_block(other), ( + f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" + ) diff --git a/tests/e2e_tests/budgets/test_soft_budget_e2e.py b/tests/e2e_tests/budgets/test_soft_budget_e2e.py new file mode 100644 index 00000000000..cd957d9dd90 --- /dev/null +++ b/tests/e2e_tests/budgets/test_soft_budget_e2e.py @@ -0,0 +1,36 @@ +"""Live e2e: soft_budget alerts but does NOT block. + +A key with a tiny `soft_budget` well under a large `max_budget`: spend crosses the +soft threshold within a couple calls, but requests keep succeeding (soft budget is +advisory). Closes the soft_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. The alert +side-effect (Slack/email) is not observable from the proxy API, so we assert the +load-bearing behavior: soft != block. +""" + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + + +def test_soft_budget_does_not_block( + client: BudgetClient, resources: ResourceManager +) -> None: + # soft far below max: spend crosses soft immediately, stays under max. + key = client.generate_key( + max_budget=1000.0, extra_params={"soft_budget": 1e-9} + ) + resources.defer(lambda: client.delete_key(key)) + + for _ in range(3): + result = client.chat( + key, "gpt-5.5", f"hi {unique_marker()}", extra_body={"max_tokens": 16} + ) + require_successful_call(result) # skip if provider unavailable + assert not is_budget_block(result), ( + "soft_budget blocked a request; it must alert only, not block " + f"(body={result.body[:200]})" + ) diff --git a/tests/e2e_tests/budgets/test_tag_budget_e2e.py b/tests/e2e_tests/budgets/test_tag_budget_e2e.py new file mode 100644 index 00000000000..5dc15720664 --- /dev/null +++ b/tests/e2e_tests/budgets/test_tag_budget_e2e.py @@ -0,0 +1,58 @@ +"""Live e2e: proxy-level tag budgets block tagged requests. + +A tag with a tiny budget: requests carrying that tag get blocked once the tag's +spend is exceeded, while a request with a different tag (no budget) still works. +Closes the proxy-level tag-budget gap in BUDGET_TEST_COVERAGE_MATRIX.md (today +only router-level tag budgets are tested). +""" + +import time + +import pytest + +from budget_client import BudgetClient, is_budget_block +from lifecycle import ResourceManager +from proxy_client import require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + +TINY_BUDGET = 1e-6 + + +def _tagged_call(client: BudgetClient, key: str, tag: str): + result = client.chat( + key, + "gpt-5.5", + f"hi {unique_marker()}", + metadata={"tags": [tag]}, + extra_body={"max_tokens": 16}, + ) + if not result.ok and not is_budget_block(result): + require_successful_call(result) # non-budget error -> skip + return result + + +def test_tag_budget_blocks_tagged_requests( + client: BudgetClient, scoped_key: str, resources: ResourceManager +) -> None: + budgeted_tag = f"e2e-budget-tag-{unique_marker()}" + client.create_tag(budgeted_tag, max_budget=TINY_BUDGET) + resources.defer(lambda: client.delete_tag(budgeted_tag)) + + # Requests under the budgeted tag get blocked once its spend is exceeded. + blocked = False + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + if is_budget_block(_tagged_call(client, scoped_key, budgeted_tag)): + blocked = True + break + time.sleep(1) + assert blocked, f"tag budget for {budgeted_tag!r} never enforced" + + # A request with an unbudgeted tag on the same key is unaffected. + free_tag = f"e2e-free-tag-{unique_marker()}" + other = _tagged_call(client, scoped_key, free_tag) + require_successful_call(other) + assert not is_budget_block(other), ( + f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget" + ) diff --git a/tests/e2e_tests/conftest.py b/tests/e2e_tests/conftest.py new file mode 100644 index 00000000000..4f73bf54786 --- /dev/null +++ b/tests/e2e_tests/conftest.py @@ -0,0 +1,54 @@ +"""Shared fixtures for all live e2e suites under tests/e2e_tests/. + +Design rule: skip on environment, fail on behavior. If the proxy is unreachable +the whole session skips; once a request reaches the proxy, behavior is asserted. + +Lifecycle: the `resources` fixture maps the init -> run -> teardown contract +(lifecycle.E2ECase) onto pytest - setup is init(), the test body is run(), and +teardown deletes every resource the test created on the long-lived proxy. + +Each suite provides its own `client` fixture (a lifecycle.ResourceClient); these +shared fixtures build on it. +""" + +from typing import Iterator + +import pytest +import requests + +from e2e_config import PROXY_BASE_URL +from lifecycle import ResourceClient, ResourceManager + + +def pytest_configure(config: pytest.Config) -> None: + config.addinivalue_line( + "markers", + "e2e: live test that requires a running proxy and real provider keys", + ) + + +@pytest.fixture(scope="session", autouse=True) +def _require_live_proxy() -> None: + """Skip the entire session unless a proxy answers its liveness probe.""" + try: + resp = requests.get(f"{PROXY_BASE_URL}/health/liveliness", timeout=5) + except requests.RequestException as exc: + pytest.skip(f"No live proxy at {PROXY_BASE_URL}: {exc}") + return + if resp.status_code >= 500: + pytest.skip(f"Proxy at {PROXY_BASE_URL} returned {resp.status_code}") + + +@pytest.fixture +def resources(client: ResourceClient) -> Iterator[ResourceManager]: + """init -> run -> teardown: create a manager, run the test, release resources.""" + manager = ResourceManager(client=client) + manager.init() + yield manager + manager.teardown() + + +@pytest.fixture +def scoped_key(resources: ResourceManager) -> str: + """A fresh all-models key per test, auto-deleted by the resources teardown.""" + return resources.key() diff --git a/tests/e2e_tests/docker-compose.yml b/tests/e2e_tests/docker-compose.yml new file mode 100644 index 00000000000..ab237f01d15 --- /dev/null +++ b/tests/e2e_tests/docker-compose.yml @@ -0,0 +1,59 @@ +# Local e2e proxy, built from the litellm_internal_staging branch. +# +# The image `litellm-staging:local` is built from a clean checkout of +# litellm_internal_staging (the working tree's uv.lock has merge markers, so +# build from a worktree, not the dirty tree): +# +# git worktree add /tmp/litellm-staging litellm_internal_staging +# docker build -t litellm-staging:local /tmp/litellm-staging +# +# Then provide provider keys (copy .env.example -> .env, fill in keys) and run: +# +# docker compose -f tests/e2e_tests/docker-compose.yml up -d +# +# The proxy comes up on http://localhost:4000 with master key sk-1234, which is +# what the e2e suites default to. Point the suites at it: +# +# LITELLM_MASTER_KEY=sk-1234 LITELLM_PROXY_URL=http://localhost:4000 \ +# .venv/bin/python -m pytest tests/e2e_tests/ -v + +services: + litellm: + image: litellm-staging:local + ports: + - "4000:4000" + command: ["--config", "/app/config.yaml", "--port", "4000"] + volumes: + - ./docker-config.yaml:/app/config.yaml + environment: + DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm" + LITELLM_MASTER_KEY: "sk-1234" + STORE_MODEL_IN_DB: "True" + env_file: + # Provider keys for the LLM-calling suites. Boots without it (those tests + # skip); copy .env.example -> .env and `docker compose up -d` to enable. + - path: .env + required: false + depends_on: + - db + - redis + + db: + image: postgres:17 + environment: + POSTGRES_DB: litellm + POSTGRES_USER: llmproxy + POSTGRES_PASSWORD: dbpassword9090 + ports: + - "5432:5432" + volumes: + - postgres_data:/var/lib/postgresql/data + + redis: + image: redis:7-alpine + ports: + - "6379:6379" + +volumes: + postgres_data: + driver: local diff --git a/tests/e2e_tests/docker-config.yaml b/tests/e2e_tests/docker-config.yaml new file mode 100644 index 00000000000..f0869c3c06b --- /dev/null +++ b/tests/e2e_tests/docker-config.yaml @@ -0,0 +1,39 @@ +# Config for the local docker e2e proxy (see docker-compose.yml). +# Minimal on purpose: just the models the e2e suites use + spend tracking + +# a redis cache (for the cache-hit spend test). Provider keys come from .env. +# Passthrough endpoints (/gemini, /anthropic) use GEMINI_API_KEY / ANTHROPIC_API_KEY +# from the environment directly, so no model_list entry is needed for them. + +general_settings: + master_key: os.environ/LITELLM_MASTER_KEY + database_url: os.environ/DATABASE_URL + store_prompts_in_spend_logs: true + +litellm_settings: + drop_params: true + cache: true + cache_params: + type: redis + host: redis + port: 6379 + +model_list: + - model_name: gpt-5.5 + litellm_params: + model: openai/gpt-5.5 + api_key: os.environ/OPENAI_API_KEY + + - model_name: claude-haiku-4-5 + litellm_params: + model: anthropic/claude-haiku-4-5 + api_key: os.environ/ANTHROPIC_API_KEY + + - model_name: gemini-2.5-flash + litellm_params: + model: gemini/gemini-2.5-flash + api_key: os.environ/GEMINI_API_KEY + + - model_name: openai-text-embedding-3-small + litellm_params: + model: openai/text-embedding-3-small + api_key: os.environ/OPENAI_API_KEY diff --git a/tests/e2e_tests/e2e_config.py b/tests/e2e_tests/e2e_config.py new file mode 100644 index 00000000000..56834d9abb0 --- /dev/null +++ b/tests/e2e_tests/e2e_config.py @@ -0,0 +1,16 @@ +"""Generic configuration for live e2e tests against a running LiteLLM proxy. + +Shared by every e2e suite under tests/e2e_tests/. Values come from the +environment so the same tests run against localhost or a deployed proxy. +""" + +import os + +PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/") +MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234") + +# Writes on the proxy are eventually consistent (e.g. spend rows flush on +# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once. +POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120")) +POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5")) +REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60")) diff --git a/tests/e2e_tests/lifecycle.py b/tests/e2e_tests/lifecycle.py new file mode 100644 index 00000000000..8130e388bff --- /dev/null +++ b/tests/e2e_tests/lifecycle.py @@ -0,0 +1,87 @@ +"""Lifecycle contract and resource cleanup for stateful e2e tests. + +Shared by every e2e suite under tests/e2e_tests/. The proxy under test is +long-lived and never reset between tests, so anything a test creates (keys, +customers, teams, orgs, users, guardrails, budgets, ...) persists unless +explicitly deleted. Every check follows an init -> run -> teardown lifecycle; +teardown releases each resource init() created, even when run() raises. + +In pytest terms (see conftest.py): the `resources` fixture's setup is init(), +the test body is run(), and the fixture's teardown is teardown(). +""" + +from dataclasses import dataclass, field +from typing import Callable, List, Protocol, runtime_checkable + + +@runtime_checkable +class E2ECase(Protocol): + """A stateful e2e check run against a long-lived proxy. + + init() acquires resources, run() exercises behaviour and asserts, teardown() + releases everything init() created. teardown() must run even if run() raises. + """ + + def init(self) -> None: ... + + def run(self) -> None: ... + + def teardown(self) -> None: ... + + +@runtime_checkable +class ResourceClient(Protocol): + """Proxy operations the convenience creators use. Resource types without a + creator here are handled generically via ResourceManager.defer().""" + + def generate_key(self, *, models: List[str]) -> str: ... + + def delete_key(self, key: str) -> None: ... + + def delete_customers(self, user_ids: List[str]) -> None: ... + + +@dataclass +class ResourceManager: + """Registry of teardown actions for resources a test creates on the stateful + proxy. + + Not limited to any resource type: register a cleanup with ``defer()`` for a + key, customer, team, org, user, guardrail, budget, MCP server - anything with + a delete. The two most common resources have sugar (``key``, ``customer``); + everything else is ``resources.defer(lambda: client.delete_team(team_id))``. + + Cleanups run LIFO (so a resource is removed before whatever it depends on) and + best-effort (one failing cleanup never blocks the rest). + """ + + client: ResourceClient + _cleanups: List[Callable[[], None]] = field( + default_factory=list + ) # mutable-ok: append-only teardown registry + + def init(self) -> None: + """No global setup needed today; present for lifecycle symmetry.""" + return None + + def defer(self, cleanup: Callable[[], None]) -> None: + """Register a teardown action for any resource the test just created.""" + self._cleanups.append(cleanup) + + def key(self) -> str: + """Create an all-models virtual key; delete it on teardown.""" + key = self.client.generate_key(models=[]) + self.defer(lambda: self.client.delete_key(key)) + return key + + def customer(self, customer_id: str) -> str: + """Track an end-user id (from the `user` param); delete it on teardown.""" + self.defer(lambda: self.client.delete_customers([customer_id])) + return customer_id + + def teardown(self) -> None: + for cleanup in reversed(self._cleanups): + try: + cleanup() + except Exception: + pass # best-effort: a failed cleanup must not block the rest diff --git a/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md b/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md new file mode 100644 index 00000000000..5e4a448857f --- /dev/null +++ b/tests/e2e_tests/llm_translation/LLM_TRANSLATION_COVERAGE_MATRIX.md @@ -0,0 +1,84 @@ +# LLM Translation Test Coverage Matrix + +Scope: the proxy's two translation surfaces, end to end against a live proxy. + +1. **Passthrough** - the client speaks the provider's NATIVE API (Gemini + `generateContent`, Anthropic `/v1/messages`); the proxy forwards it and still + logs a costed `SpendLogs` row (`call_type="pass_through_endpoint"`). Routes: + `/gemini`, `/anthropic`, `/vertex_ai`, `/openai`, `/bedrock`, `/cohere`, + `/mistral`, `/vllm`. +2. **Non-passthrough** - the client speaks OpenAI format + (`/chat/completions`, `/embeddings`); litellm translates to/from the provider. + +The two axes that must work in production for each: **passthrough vs +non-passthrough** and **streaming vs non-streaming**, with **cost logged** and +**tool calls** working in every cell. + +Companion: live suite `test_passthrough_e2e.py` (this directory). The +non-passthrough chat/embedding cells are exercised by `../spend_tracking/`. + +Levels: `live` real provider + proxy + SpendLogs row; `unit` mocked. +Status: `covered` / `partial` / `gap`. + +--- + +## Passthrough endpoints (native provider format) + +| Provider | Non-streaming | Streaming | Tool calls | Cost logged | Status | +|----------|---------------|-----------|------------|-------------|--------| +| Gemini (`/gemini/v1beta/models/{m}:generateContent` / `:streamGenerateContent`) | live | live | live | live | **covered** | +| Anthropic (`/anthropic/v1/messages`) | live | live | live | live | **covered** | +| Vertex AI (`/vertex_ai/...`) | - | - | - | - | gap (gcloud auth) | +| OpenAI / Bedrock / Cohere / Mistral / VLLM | - | - | - | - | gap | + +Each covered cell asserts: `call_type == "pass_through_endpoint"`, `spend > 0`, +`status == "success"`, correct `custom_llm_provider`/`model`, row correlated by the +`x-litellm-call-id` header. Gemini non-streaming also pins `request_tags` +propagation; streaming pins `chunks > 0` then a costed row; tool tests assert the +provider emitted a tool call (`functionCall` / `tool_use`) and it was costed. + +Cost on passthrough is computed in the success handler by transforming the native +response to a `ModelResponse` and calling `litellm.completion_cost()`; for +streaming, chunks are buffered and costed after the stream ends. This is the path +most likely to silently break and the one a mock can't prove works. + +## Non-passthrough endpoints (OpenAI-compatible translation) + +| Modality | Non-streaming | Streaming | Tool calls | Cost logged | Status | +|----------|---------------|-----------|------------|-------------|--------| +| Chat | live (spend suite) | live (spend suite) | gap | live | partial | +| Embeddings | live (spend suite) | n/a | n/a | live | covered | +| Responses / image / audio / rerank / realtime | - | - | - | - | gap | + +## This suite's files + +| Test | Cell | +|------|------| +| `test_gemini_passthrough_nonstreaming_logs_cost` | gemini native, non-stream, cost + tags | +| `test_gemini_passthrough_streaming_logs_cost` | gemini native, stream, cost | +| `test_gemini_passthrough_tool_call_logs_cost` | gemini native, tool call, cost | +| `test_anthropic_passthrough_nonstreaming_logs_cost` | anthropic native, non-stream, cost | +| `test_anthropic_passthrough_streaming_logs_cost` | anthropic native, stream, cost | +| `test_anthropic_passthrough_tool_call_logs_cost` | anthropic native, tool call, cost | + +## Gaps + +- Vertex / OpenAI / Bedrock / Cohere passthrough (same shape; add once the + provider credential is configured; Vertex is closest - route exists, auth stale). +- Non-passthrough tool calls over `/chat/completions` end to end with cost. +- Image / audio / rerank / responses / realtime translation + cost. +- Streaming cost-injection (`include_cost_in_streaming_usage`); passthrough on + client disconnect (partial-usage logging). + +## Adding a provider/modality + +Extend `PassthroughClient` with the native call (it inherits keys, cleanup, and +SpendLogs polling from `ProxyClient`), then add a test that calls it, +`require_successful_call(result)`, and `_costed_row(...)`. + +## Timing + +Passthrough spend is logged asynchronously after the response and lands on the +`proxy_batch_write_at` (~60s) cycle, so cost assertions poll +`/spend/logs?request_id=` to a deadline. Streaming cost is only +known after the stream is fully consumed. diff --git a/tests/e2e_tests/llm_translation/conftest.py b/tests/e2e_tests/llm_translation/conftest.py new file mode 100644 index 00000000000..58f2b3e4dc1 --- /dev/null +++ b/tests/e2e_tests/llm_translation/conftest.py @@ -0,0 +1,16 @@ +"""LLM-translation suite's `client` fixture. + +The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker +live in the parent tests/e2e_tests/conftest.py. PassthroughClient subclasses +ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources` +fixture cleans up keys this suite creates. +""" + +import pytest + +from passthrough_client import PassthroughClient, build_client + + +@pytest.fixture(scope="session") +def client() -> PassthroughClient: + return build_client() diff --git a/tests/e2e_tests/llm_translation/passthrough_client.py b/tests/e2e_tests/llm_translation/passthrough_client.py new file mode 100644 index 00000000000..458f31edff8 --- /dev/null +++ b/tests/e2e_tests/llm_translation/passthrough_client.py @@ -0,0 +1,136 @@ +"""Client for LLM-translation e2e tests over the proxy's passthrough endpoints. + +Extends the shared ProxyClient with native provider passthrough calls. A +passthrough request is sent in the PROVIDER's native format (Gemini +generateContent, Anthropic /v1/messages) to the proxy, which forwards it to the +provider and still logs a SpendLogs row (call_type="pass_through_endpoint"). The +litellm virtual key is passed as the provider key; the proxy swaps in the real +env credential. SpendLogs.request_id == the x-litellm-call-id response header. +""" + +from dataclasses import dataclass +from typing import List, Optional + +import requests + +from proxy_client import ProxyClient, proxy_client_kwargs + + +@dataclass(frozen=True, slots=True) +class PassthroughResult: + """Outcome of a native passthrough call. ``call_id`` correlates to the row.""" + + status_code: int + call_id: Optional[str] # x-litellm-call-id -> SpendLogs.request_id + body: str + chunks: int = 0 # number of streamed events (0 for non-streaming) + + @property + def ok(self) -> bool: + return 200 <= self.status_code < 300 + + +def _tag_header(tags: Optional[List[str]]) -> dict: + return {"tags": ",".join(tags)} if tags else {} + + +class PassthroughClient(ProxyClient): + # ---- Gemini native passthrough (/gemini/v1beta/...) ----------------- + + def gemini_generate( + self, + key: str, + model: str, + text: str, + *, + tools: Optional[list] = None, + tags: Optional[List[str]] = None, + ) -> PassthroughResult: + body: dict = {"contents": [{"role": "user", "parts": [{"text": text}]}]} + if tools is not None: + body["tools"] = tools + headers = { + "x-goog-api-key": key, + "Content-Type": "application/json", + **_tag_header(tags), + } + resp = requests.post( + f"{self._base_url}/gemini/v1beta/models/{model}:generateContent", + headers=headers, + json=body, + timeout=self._request_timeout, + ) + return PassthroughResult( + resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text + ) + + def gemini_stream( + self, key: str, model: str, text: str, *, tags: Optional[List[str]] = None + ) -> PassthroughResult: + headers = { + "x-goog-api-key": key, + "Content-Type": "application/json", + **_tag_header(tags), + } + resp = requests.post( + f"{self._base_url}/gemini/v1beta/models/{model}:streamGenerateContent", + headers=headers, + params={"alt": "sse"}, + json={"contents": [{"role": "user", "parts": [{"text": text}]}]}, + stream=True, + timeout=self._request_timeout, + ) + call_id = resp.headers.get("x-litellm-call-id") + if not (200 <= resp.status_code < 300): + return PassthroughResult(resp.status_code, call_id, resp.text) + chunks = sum(1 for line in resp.iter_lines() if line) + return PassthroughResult(resp.status_code, call_id, "", chunks) + + # ---- Anthropic native passthrough (/anthropic/v1/messages) ---------- + + def anthropic_message( + self, + key: str, + model: str, + text: str, + *, + max_tokens: int = 64, + tools: Optional[list] = None, + stream: bool = False, + tags: Optional[List[str]] = None, + ) -> PassthroughResult: + body: dict = { + "model": model, + "max_tokens": max_tokens, + "messages": [{"role": "user", "content": text}], + } + if tools is not None: + body["tools"] = tools + if stream: + body["stream"] = True + headers = { + "x-api-key": key, + "anthropic-version": "2023-06-01", + "Content-Type": "application/json", + **_tag_header(tags), + } + url = f"{self._base_url}/anthropic/v1/messages" + if not stream: + resp = requests.post( + url, headers=headers, json=body, timeout=self._request_timeout + ) + return PassthroughResult( + resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text + ) + resp = requests.post( + url, headers=headers, json=body, stream=True, timeout=self._request_timeout + ) + call_id = resp.headers.get("x-litellm-call-id") + if not (200 <= resp.status_code < 300): + return PassthroughResult(resp.status_code, call_id, resp.text) + chunks = sum(1 for line in resp.iter_lines() if line) + return PassthroughResult(resp.status_code, call_id, "", chunks) + + +def build_client() -> PassthroughClient: + return PassthroughClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/llm_translation/test_passthrough_e2e.py b/tests/e2e_tests/llm_translation/test_passthrough_e2e.py new file mode 100644 index 00000000000..b7250898200 --- /dev/null +++ b/tests/e2e_tests/llm_translation/test_passthrough_e2e.py @@ -0,0 +1,158 @@ +"""Live e2e for LLM-translation passthrough endpoints. + +Each test sends a NATIVE provider request through the proxy's passthrough route +and verifies the proxy still logged a costed SpendLogs row +(call_type="pass_through_endpoint"), correlated by the x-litellm-call-id header. + +Covered: gemini ("gemini-2.5-flash") + anthropic ("claude-haiku-4-5"), streaming + +non-streaming, plus native tool calls. See LLM_TRANSLATION_COVERAGE_MATRIX.md. + +Skip on environment, fail on behavior: a passthrough call returning non-2xx +(provider key missing) skips; once it returns 2xx, a missing/zero-cost row fails. +""" + +import pytest + +from passthrough_client import PassthroughClient, PassthroughResult +from proxy_client import SpendLogRow, require_successful_call, unique_marker + +pytestmark = pytest.mark.e2e + + +def _f(value: object) -> float: + return float(value) if value is not None else 0.0 # type: ignore[arg-type] + + +def _s(value: object) -> str: + return str(value) if value is not None else "" + + +def _costed_row(client: PassthroughClient, result: PassthroughResult) -> SpendLogRow: + """The passthrough call's logged row, polled until it carries a cost. + + Asserts (not skips) that a 2xx passthrough call produced a costed row - the + whole point of passthrough spend tracking. + """ + assert result.call_id, "passthrough response had no x-litellm-call-id header" + rows = client.poll_logs_for_request_id( + result.call_id, + predicate=lambda rs: _f(rs[0].get("spend")) > 0, + ) + assert rows, f"no SpendLogs row for passthrough call_id {result.call_id}" + row = rows[0] + assert _s(row.get("call_type")) == "pass_through_endpoint" + assert _f(row.get("spend")) > 0, f"passthrough call was not costed: {row}" + assert _s(row.get("status")) == "success" + return row + + +# ---- Gemini passthrough ------------------------------------------------ + + +def test_gemini_passthrough_nonstreaming_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + tag = f"e2e-passthrough-{unique_marker()}" + result = client.gemini_generate( + scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"] + ) + require_successful_call(result) + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "gemini" + assert "gemini" in _s(row.get("model")) + assert tag in _s(row.get("request_tags")), f"tags not logged: {row.get('request_tags')}" + + +def test_gemini_passthrough_streaming_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five") + require_successful_call(result) + assert result.chunks > 0, "streaming passthrough produced no events" + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "gemini" + + +def test_gemini_passthrough_tool_call_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + result = client.gemini_generate( + scoped_key, + "gemini-2.5-flash", + "What is the weather in Paris? Use the get_weather tool.", + tools=[ + { + "functionDeclarations": [ + { + "name": "get_weather", + "description": "Get the weather for a city", + "parameters": { + "type": "object", + "properties": {"city": {"type": "string"}}, + "required": ["city"], + }, + } + ] + } + ], + ) + require_successful_call(result) + assert "functionCall" in result.body, "gemini did not emit a tool call" + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "gemini" + + +# ---- Anthropic passthrough --------------------------------------------- + + +def test_anthropic_passthrough_nonstreaming_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello") + require_successful_call(result) + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "anthropic" + assert "claude" in _s(row.get("model")) + + +def test_anthropic_passthrough_streaming_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + result = client.anthropic_message( + scoped_key, "claude-haiku-4-5", "Count to five", stream=True + ) + require_successful_call(result) + assert result.chunks > 0, "streaming passthrough produced no events" + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "anthropic" + + +def test_anthropic_passthrough_tool_call_logs_cost( + client: PassthroughClient, scoped_key: str +) -> None: + result = client.anthropic_message( + scoped_key, + "claude-haiku-4-5", + "What is the weather in Paris? Use the get_weather tool.", + tools=[ + { + "name": "get_weather", + "description": "Get the weather for a city", + "input_schema": { + "type": "object", + "properties": {"city": {"type": "string"}}, + "required": ["city"], + }, + } + ], + ) + require_successful_call(result) + assert "tool_use" in result.body, "anthropic did not emit a tool call" + + row = _costed_row(client, result) + assert _s(row.get("custom_llm_provider")) == "anthropic" diff --git a/tests/e2e_tests/proxy_client.py b/tests/e2e_tests/proxy_client.py new file mode 100644 index 00000000000..deeaf830f2c --- /dev/null +++ b/tests/e2e_tests/proxy_client.py @@ -0,0 +1,371 @@ +"""Generic HTTP client for live e2e tests against a running LiteLLM proxy. + +Shared by every e2e suite under tests/e2e_tests/. Covers the proxy operations any +suite needs: key/customer management (so the shared ResourceManager can clean up), +OpenAI-compatible calls, route probing, and SpendLogs read-back. Suite-specific +clients subclass ProxyClient (see spend_tracking/, llm_translation/, budgets/). + +Talks to the proxy over real HTTP so every test sees what a real client sees: +the x-litellm-call-id header, the response body, and the rows the proxy writes. +Writes are eventually consistent (proxy_batch_write_at ~60s), so read-backs poll +to a deadline rather than sleeping once. +""" + +import json +import time +import uuid +from dataclasses import dataclass +from typing import Callable, Dict, List, Optional, Protocol, runtime_checkable + +import pytest +import requests + +from e2e_config import ( + MASTER_KEY, + POLL_INTERVAL, + POLL_TIMEOUT, + PROXY_BASE_URL, + REQUEST_TIMEOUT, +) + +SpendLogRow = Dict[str, object] + + +@runtime_checkable +class CallOutcome(Protocol): + """Anything with an HTTP status and body that can pass the skip/fail boundary.""" + + status_code: int + body: str + + @property + def ok(self) -> bool: ... + + +@dataclass(frozen=True, slots=True) +class ProbeResult: + """Outcome of a single route probe: enough to see *why* it (mis)behaved.""" + + url: str + status_code: int + body: str + + @property + def healthy(self) -> bool: + # Route exists (not 404), handler did not crash (not 5xx), request + # completed (not -1). A 4xx (missing params/auth) still means it ran. + return 200 <= self.status_code != 404 and self.status_code < 500 + + def __str__(self) -> str: + return f"GET {self.url} -> {self.status_code}\n{self.body[:600]}" + + +@dataclass(frozen=True, slots=True) +class CallResult: + """Outcome of a single OpenAI-compatible call made through the proxy.""" + + status_code: int + call_id: Optional[str] # x-litellm-call-id response header + response_id: Optional[str] # body "id"; SpendLogs.request_id is derived from this + response_cost_header: Optional[str] # x-litellm-response-cost header + body: str + content: Optional[str] + + @property + def ok(self) -> bool: + return 200 <= self.status_code < 300 + + +def _auth(key: str) -> Dict[str, str]: + return {"Authorization": f"Bearer {key}", "Content-Type": "application/json"} + + +class ProxyClient: + def __init__( + self, + base_url: str, + master_key: str, + *, + request_timeout: float, + poll_timeout: float, + poll_interval: float, + ) -> None: + self._base_url = base_url.rstrip("/") + self._master_key = master_key + self._request_timeout = request_timeout + self._poll_timeout = poll_timeout + self._poll_interval = poll_interval + + # ---- key / customer management (satisfies lifecycle.ResourceClient) ---- + + def generate_key( + self, + *, + models: Optional[List[str]] = None, + max_budget: Optional[float] = None, + metadata: Optional[Dict[str, object]] = None, + extra_params: Optional[Dict[str, object]] = None, + ) -> str: + payload: Dict[str, object] = {"models": models or [], "duration": None} + if max_budget is not None: + payload["max_budget"] = max_budget + if metadata is not None: + payload["metadata"] = metadata + if extra_params: + payload.update(extra_params) + resp = requests.post( + f"{self._base_url}/key/generate", + headers=_auth(self._master_key), + json=payload, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return str(resp.json()["key"]) + + def key_info(self, key: str) -> Dict[str, object]: + resp = requests.get( + f"{self._base_url}/key/info", + headers=_auth(self._master_key), + params={"key": key}, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return dict(resp.json().get("info", {})) + + def delete_key(self, key: str) -> None: + """Best-effort teardown; a failed cleanup must not fail the test.""" + try: + requests.post( + f"{self._base_url}/key/delete", + headers=_auth(self._master_key), + json={"keys": [key]}, + timeout=self._request_timeout, + ) + except requests.RequestException: + pass + + def delete_customers(self, user_ids: List[str]) -> None: + """Best-effort teardown of end-user/customer rows the `user` param creates.""" + if not user_ids: + return + try: + requests.post( + f"{self._base_url}/customer/delete", + headers=_auth(self._master_key), + json={"user_ids": user_ids}, + timeout=self._request_timeout, + ) + except requests.RequestException: + pass + + # ---- OpenAI-compatible calls ---------------------------------------- + + def chat( + self, + key: str, + model: str, + content: str, + *, + stream: bool = False, + metadata: Optional[Dict[str, object]] = None, + extra_body: Optional[Dict[str, object]] = None, + ) -> CallResult: + body: Dict[str, object] = { + "model": model, + "messages": [{"role": "user", "content": content}], + "stream": stream, + } + if metadata is not None: + body["metadata"] = metadata + if extra_body is not None: + body.update(extra_body) + url = f"{self._base_url}/chat/completions" + if stream: + return self._chat_stream(url, key, body) + resp = requests.post( + url, headers=_auth(key), json=body, timeout=self._request_timeout + ) + parsed = resp.json() if resp.content else {} + choices = parsed.get("choices") or [{}] + message_content = (choices[0].get("message") or {}).get("content") + return CallResult( + status_code=resp.status_code, + call_id=resp.headers.get("x-litellm-call-id"), + response_id=parsed.get("id"), + response_cost_header=resp.headers.get("x-litellm-response-cost"), + body=resp.text, + content=message_content, + ) + + def _chat_stream(self, url: str, key: str, body: Dict[str, object]) -> CallResult: + resp = requests.post( + url, + headers=_auth(key), + json=body, + stream=True, + timeout=self._request_timeout, + ) + if not (200 <= resp.status_code < 300): + return CallResult( + status_code=resp.status_code, + call_id=resp.headers.get("x-litellm-call-id"), + response_id=None, + response_cost_header=resp.headers.get("x-litellm-response-cost"), + body=resp.text, + content=None, + ) + response_id: Optional[str] = None + parts: List[str] = [] + for raw in resp.iter_lines(): + if not raw: + continue + line = raw.decode("utf-8") + if not line.startswith("data:"): + continue + data = line[len("data:") :].strip() + if data == "[DONE]": + break + chunk = json.loads(data) + response_id = chunk.get("id", response_id) + for choice in chunk.get("choices", []): + piece = (choice.get("delta") or {}).get("content") + if piece: + parts.append(piece) + return CallResult( + status_code=resp.status_code, + call_id=resp.headers.get("x-litellm-call-id"), + response_id=response_id, + response_cost_header=resp.headers.get("x-litellm-response-cost"), + body="", + content="".join(parts) or None, + ) + + def embed(self, key: str, model: str, text: str) -> CallResult: + resp = requests.post( + f"{self._base_url}/embeddings", + headers=_auth(key), + json={"model": model, "input": text}, + timeout=self._request_timeout, + ) + parsed = resp.json() if resp.content else {} + return CallResult( + status_code=resp.status_code, + call_id=resp.headers.get("x-litellm-call-id"), + response_id=parsed.get("id"), + response_cost_header=resp.headers.get("x-litellm-response-cost"), + body=resp.text, + content=None, + ) + + # ---- route discovery ------------------------------------------------- + + def get_openapi(self) -> Dict[str, object]: + """The proxy's live route schema from /openapi.json.""" + resp = requests.get( + f"{self._base_url}/openapi.json", timeout=self._request_timeout + ) + resp.raise_for_status() + return dict(resp.json()) + + def probe(self, path: str, params: Optional[Dict[str, str]] = None) -> ProbeResult: + """GET a route with master-key auth; capture status + body to show why.""" + url = f"{self._base_url}{path}" + try: + resp = requests.get( + url, + headers=_auth(self._master_key), + params=params or {}, + timeout=self._request_timeout, + ) + except requests.RequestException as exc: + return ProbeResult(url=url, status_code=-1, body=f"request error: {exc}") + return ProbeResult(url=url, status_code=resp.status_code, body=resp.text) + + # ---- SpendLogs read-back -------------------------------------------- + + def _get_logs( + self, *, request_id: Optional[str] = None, api_key: Optional[str] = None + ) -> List[SpendLogRow]: + params: Dict[str, str] = {} + if request_id is not None: + params["request_id"] = request_id + if api_key is not None: + params["api_key"] = api_key + resp = requests.get( + f"{self._base_url}/spend/logs", + headers=_auth(self._master_key), + params=params, + timeout=self._request_timeout, + ) + if resp.status_code != 200: + return [] + data = resp.json() + return [dict(row) for row in data] if isinstance(data, list) else [] + + def poll_logs_for_key( + self, + key: str, + *, + min_rows: int = 1, + predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None, + ) -> List[SpendLogRow]: + return self._poll(lambda: self._get_logs(api_key=key), min_rows, predicate) + + def poll_logs_for_request_id( + self, + request_id: str, + *, + min_rows: int = 1, + predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None, + ) -> List[SpendLogRow]: + return self._poll( + lambda: self._get_logs(request_id=request_id), min_rows, predicate + ) + + def _poll( + self, + fetch: Callable[[], List[SpendLogRow]], + min_rows: int, + predicate: Optional[Callable[[List[SpendLogRow]], bool]], + ) -> List[SpendLogRow]: + deadline = time.monotonic() + self._poll_timeout + rows: List[SpendLogRow] = [] + while time.monotonic() < deadline: + rows = fetch() + satisfied = len(rows) >= min_rows and ( + predicate is None or predicate(rows) + ) + if satisfied: + return rows + time.sleep(self._poll_interval) + return rows + + +def proxy_client_kwargs() -> Dict[str, object]: + """Constructor kwargs shared by every ProxyClient subclass.""" + return { + "base_url": PROXY_BASE_URL, + "master_key": MASTER_KEY, + "request_timeout": REQUEST_TIMEOUT, + "poll_timeout": POLL_TIMEOUT, + "poll_interval": POLL_INTERVAL, + } + + +def require_successful_call(result: CallOutcome) -> None: + """The skip/fail boundary. + + A non-2xx here means the environment cannot make the call (missing provider + key, upstream down) -> skip. Everything downstream of a 2xx is behavior that + is allowed to fail. + """ + if result.ok: + return + pytest.skip( + f"upstream call unavailable (status {result.status_code}); " + f"body={result.body[:300]}" + ) + + +def unique_marker() -> str: + return uuid.uuid4().hex[:12] diff --git a/tests/e2e_tests/pytest.ini b/tests/e2e_tests/pytest.ini new file mode 100644 index 00000000000..7682dab96ed --- /dev/null +++ b/tests/e2e_tests/pytest.ini @@ -0,0 +1,7 @@ +[pytest] +# Config when any e2e suite under tests/e2e_tests/ is run directly, e.g. +# uv run pytest tests/e2e_tests/spend_tracking/ -v +# The e2e marker is also registered in conftest.py for runs rooted elsewhere. +addopts = --strict-markers --strict-config +markers = + e2e: live test that requires a running proxy and real provider keys diff --git a/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md b/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md new file mode 100644 index 00000000000..fbb5ce60c51 --- /dev/null +++ b/tests/e2e_tests/spend_tracking/SPEND_TRACKING_COVERAGE_MATRIX.md @@ -0,0 +1,77 @@ +# Spend Tracking Test Coverage Matrix + +Scope: every distinct spend-tracking code path, mapped to the test that exercises +it and the level it runs at. Highlights where a live e2e check is the only thing +that would catch a regression. + +Companion: live suite `test_spend_tracking_e2e.py` + route breadth +`test_spend_routes.py` (this directory). Offline regression suite: +`tests/test_litellm/proxy/spend_tracking/`. Reference PR: BerriAI/litellm#29956. + +Levels: `unit` mocked; `integration` real DB/cost-map; `live` real provider + +proxy + SpendLogs rows. Status: `covered` / `partial` / `gap`. + +--- + +## SpendLogs row construction (`spend_tracking_utils.get_logging_payload`) + +| Path | Existing | Level | Status | Live e2e | +|------|----------|-------|--------|----------| +| `_get_status_for_spend_log` | `test_spend_tracking_utils.py` | unit | covered | yes (status read off the row) | +| cache-hit `request_id` suffix | `test_spend_tracking_utils.py` | unit | covered | yes (`test_cache_hit_is_zero_cost_and_suffixed`) | +| failure status + zero spend | `test_spend_tracking_utils.py` | unit | partial | yes (`test_failure_call_writes_failure_status_row`) | +| field population (model/tokens/api_key/team/org) | `test_spend_tracking_utils.py` | unit | partial | yes (asserts real values) | +| `request_tags` propagation | `test_db_spend_update_writer.py` | unit | partial | yes (`test_request_tags_round_trip`) | +| `end_user` attribution | unit | unit | partial | yes (`test_end_user_spend_attributed_on_row`) | + +## Cost calculation by modality + +| Modality | Existing | Status | Live e2e | +|----------|----------|--------|----------| +| Chat (non-stream) | `test_cost_calculator.py`, `local_testing/test_completion_cost.py` | covered | yes (`test_chat_completion_writes_nonzero_spend_row`) | +| Chat (streaming) | `test_streaming_interrupt_spend_tracking.py` | partial | yes (`test_streaming_chat_completion_tracks_spend`) | +| Embedding | `test_cost_calculator.py` (#29956) | partial | yes (`test_embedding_writes_nonzero_spend_row`) | +| Pass-through (gemini/anthropic) | `pass_through_tests/*.test.js` + `llm_translation/` suite | covered | yes (llm_translation suite) | +| Image / audio / rerank / responses / realtime | per-provider unit cost tests | partial/gap | gap | + +## Entity spend aggregation + +| Entity | Existing | Status | Live e2e | +|--------|----------|--------|----------| +| API key | `test_db_spend_update_writer.py`, `test_spend_counters.py` | covered | yes (`test_key_spend_equals_sum_of_logs`) | +| Tag | `test_update_daily_tag_spend.py` | partial | yes (`test_tag_spend_matches_sum_of_tagged_logs`) | +| End-user | `test_proxy_update_spend.py` | covered | yes | +| Spend == sum(logs) consistency | none | gap | yes (key + tag aggregate == sum of rows) | + +## Spend read endpoints (verification surface) + +| Endpoint | Existing | Status | Live e2e | +|----------|----------|--------|----------| +| `/spend/logs` (request_id / api_key) | `test_spend_management_endpoints.py` | covered | yes (primary read path) | +| `/spend/calculate` | `local_testing/test_spend_calculate_endpoint.py` | covered | yes (`test_spend_calculate_returns_nonzero_cost`) | +| `/spend/tags` | `test_spend_management_endpoints.py` | partial | yes (tag accuracy test) | +| whole spend GET surface (22 routes) | unit per-handler | partial | yes (`test_spend_routes.py` probes each for 404/5xx) | + +## What this suite pins + +| Test | Invariant | +|------|-----------| +| `test_chat_completion_writes_nonzero_spend_row` | nonzero cost, token arithmetic, status, row findable by `response.id` | +| `test_streaming_chat_completion_tracks_spend` | streamed responses still costed | +| `test_embedding_writes_nonzero_spend_row` | embedding cost, `completion_tokens == 0` | +| `test_cache_hit_is_zero_cost_and_suffixed` | cache hits not double-charged; `_cache_hit` suffix | +| `test_key_spend_equals_sum_of_logs` | key aggregate == sum of rows | +| `test_request_tags_round_trip` | tags persist onto the row | +| `test_tag_spend_matches_sum_of_tagged_logs` | `/spend/tags` SUM/COUNT == tagged rows | +| `test_end_user_spend_attributed_on_row` | `end_user` attributed + costed | +| `test_failure_call_writes_failure_status_row` | failed call -> `status=failure`, `spend=0` | +| `test_spend_calculate_returns_nonzero_cost` | cost-map smoke (no batch wait) | +| `test_spend_routes.py` (23) | no spend route 404s or 5xxs | + +## Design + timing + +`proxy_batch_write_at` (~60s) means rows land late; every read polls to a deadline. +Fresh scoped key per test (isolation, xdist-safe, cleaned up). Assert invariants +(`spend > 0`, `total == prompt + completion`, aggregate == sum), not literal +$/token values, so pricing drift is not a failure. Skip on environment (no proxy / +no provider key), fail on behavior (a real 2xx call with a wrong/missing row). diff --git a/tests/e2e_tests/spend_tracking/conftest.py b/tests/e2e_tests/spend_tracking/conftest.py new file mode 100644 index 00000000000..d0e82691489 --- /dev/null +++ b/tests/e2e_tests/spend_tracking/conftest.py @@ -0,0 +1,16 @@ +"""Spend-tracking suite's `client` fixture. + +The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker +live in the parent tests/e2e_tests/conftest.py. SpendE2EClient satisfies the +lifecycle.ResourceClient protocol, so the shared `resources` fixture cleans up +keys and customers this suite creates. +""" + +import pytest + +from spend_e2e_client import SpendE2EClient, build_client + + +@pytest.fixture(scope="session") +def client() -> SpendE2EClient: + return build_client() diff --git a/tests/e2e_tests/spend_tracking/spend_e2e_client.py b/tests/e2e_tests/spend_tracking/spend_e2e_client.py new file mode 100644 index 00000000000..f7a1451eb8a --- /dev/null +++ b/tests/e2e_tests/spend_tracking/spend_e2e_client.py @@ -0,0 +1,94 @@ +"""Spend-tracking e2e client: the generic ProxyClient plus spend-specific reads. + +Generic proxy operations (keys, customers, chat/embed, route probing, SpendLogs +polling) live in the shared tests/e2e_tests/proxy_client.py. This module adds only +the spend-specific endpoints: /spend/calculate, /spend/tags, and key-spend polling. + +Re-exports CallResult / SpendLogRow / require_successful_call / unique_marker so +existing tests keep importing them from here. +""" + +import time +from typing import List, Optional + +import requests + +from proxy_client import ( + CallResult, + ProbeResult, + ProxyClient, + SpendLogRow, + _auth, + proxy_client_kwargs, + require_successful_call, + unique_marker, +) + +__all__ = [ + "SpendE2EClient", + "build_client", + "CallResult", + "ProbeResult", + "SpendLogRow", + "require_successful_call", + "unique_marker", +] + + +class SpendE2EClient(ProxyClient): + def calculate_spend(self, model: str, content: str) -> float: + resp = requests.post( + f"{self._base_url}/spend/calculate", + headers=_auth(self._master_key), + json={"model": model, "messages": [{"role": "user", "content": content}]}, + timeout=self._request_timeout, + ) + resp.raise_for_status() + return float(resp.json()["cost"]) + + def spend_by_tags(self) -> List[SpendLogRow]: + """/spend/tags: SUM(spend) GROUP BY tag, straight from SpendLogs.""" + resp = requests.get( + f"{self._base_url}/spend/tags", + headers=_auth(self._master_key), + timeout=self._request_timeout, + ) + if resp.status_code != 200: + return [] + data = resp.json() + if isinstance(data, dict) and "spend_per_tag" in data: + data = data["spend_per_tag"] + return [dict(row) for row in data] if isinstance(data, list) else [] + + def poll_tag_spend( + self, tag: str, *, minimum: float = 0.0 + ) -> Optional[SpendLogRow]: + """Poll until the tag's aggregate spend reaches `minimum`; last seen entry.""" + deadline = time.monotonic() + self._poll_timeout + entry: Optional[SpendLogRow] = None + while time.monotonic() < deadline: + matches = [ + row + for row in self.spend_by_tags() + if str(row.get("individual_request_tag")) == tag + ] + if matches: + entry = matches[0] + if float(entry.get("total_spend") or 0.0) >= minimum: + return entry + time.sleep(self._poll_interval) + return entry + + def poll_key_spend(self, key: str, *, minimum: float = 0.0) -> float: + deadline = time.monotonic() + self._poll_timeout + spend = 0.0 + while time.monotonic() < deadline: + spend = float(self.key_info(key).get("spend") or 0.0) + if spend > minimum: + return spend + time.sleep(self._poll_interval) + return spend + + +def build_client() -> SpendE2EClient: + return SpendE2EClient(**proxy_client_kwargs()) diff --git a/tests/e2e_tests/spend_tracking/test_spend_routes.py b/tests/e2e_tests/spend_tracking/test_spend_routes.py new file mode 100644 index 00000000000..f755eeea0b0 --- /dev/null +++ b/tests/e2e_tests/spend_tracking/test_spend_routes.py @@ -0,0 +1,93 @@ +"""Breadth check: query every route on the spend read surface and show what it +returns. + +Spend tracking sprawls across many routes (model-cost / key / user / team / org / +customer aggregation, tags, and activity reports). Most are served with +`include_in_schema=False`, so they do NOT appear in `/openapi.json` - discovery +from the schema alone misses ~70% of the surface. So we probe a curated, verified +list directly, plus any spend route the schema does list (to auto-catch new ones). + +Each probe captures status AND body, so a failure shows the proxy's actual error +(a 500 traceback, a 404 meaning the route was removed) rather than a bare code. +Run with `-rA` (or `-s`) to print every route's response, not just failures. + +Healthy == route exists (not 404) and handler did not crash (not 5xx). A 4xx +(missing params / auth nuance) still means the route is wired and ran. Cheap and +fast: no batch-write wait, no provider calls. +""" + +from datetime import datetime, timedelta, timezone +from typing import Dict, List + +import pytest + +from spend_e2e_client import SpendE2EClient + +pytestmark = pytest.mark.e2e + +# Verified present and responsive on a live proxy. One per row of the spend +# surface: key / user / team / org / customer aggregation, model-cost, tags, +# activity. +SPEND_ROUTES = ( + "/spend/keys", + "/spend/users", + "/spend/tags", + "/spend/logs", + "/spend/logs/ui", + "/global/spend", + "/global/spend/keys", + "/global/spend/teams", + "/global/spend/models", + "/global/spend/provider", + "/global/spend/report", + "/global/spend/tags", + "/global/spend/logs", + "/global/spend/all_tag_names", + "/global/activity", + "/global/activity/model", + "/global/activity/exceptions", + "/key/list", + "/user/list", + "/team/list", + "/organization/list", + "/customer/list", +) + +_SPEND_PREFIXES = ("/spend", "/global/spend", "/global/activity") + + +def _default_params() -> Dict[str, str]: + # Satisfies date-required endpoints (report/activity/provider); ignored elsewhere. + end = datetime.now(timezone.utc).date() + start = end - timedelta(days=1) + return {"start_date": start.isoformat(), "end_date": end.isoformat()} + + +@pytest.mark.parametrize("route", SPEND_ROUTES) +def test_spend_route_responsive(client: SpendE2EClient, route: str) -> None: + result = client.probe(route, params=_default_params()) + print(result) # shown on failure, and for all routes under `-rA` / `-s` + assert result.healthy, str(result) + + +def test_schema_listed_spend_routes_are_responsive(client: SpendE2EClient) -> None: + """Probe any spend GET route the schema lists that isn't in SPEND_ROUTES.""" + paths = client.get_openapi().get("paths", {}) + assert isinstance(paths, dict) and paths, "/openapi.json had no paths" + + discovered: List[str] = [ + path + for path, operations in paths.items() + if isinstance(operations, dict) + and "get" in {m.lower() for m in operations} + and "{" not in path + and any(path.startswith(prefix) for prefix in _SPEND_PREFIXES) + ] + extras = [path for path in discovered if path not in SPEND_ROUTES] + + params = _default_params() + results = [client.probe(path, params=params) for path in extras] + for result in results: + print(result) + offenders = [str(result) for result in results if not result.healthy] + assert not offenders, "non-responsive schema spend routes:\n" + "\n".join(offenders) diff --git a/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py b/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py new file mode 100644 index 00000000000..142eb4da40f --- /dev/null +++ b/tests/e2e_tests/spend_tracking/test_spend_tracking_e2e.py @@ -0,0 +1,335 @@ +"""Live end-to-end spend-tracking tests against a running proxy. + +Run against a proxy started with tests/e2e_tests/docker-config.yaml (or the +gateway config). Coverage rationale: SPEND_TRACKING_COVERAGE_MATRIX.md. + +Model names are literals from that config: chat tests hit "gemini-2.5-flash", +embedding tests hit "openai-text-embedding-3-small". + +Every test: fresh scoped key (isolation) -> real provider call -> +require_successful_call (skip iff env can't make it) -> poll /spend/logs to a +deadline (rows land ~60s later via proxy_batch_write_at) -> assert invariants on +the real row (spend, token arithmetic, status, cache). + +Assertions target invariants, not literals: a regression in the spend pipeline +fails the test; a pricing or token-count drift does not. +""" + +from typing import Callable, Dict, List + +import pytest + +from lifecycle import ResourceManager +from spend_e2e_client import ( + SpendE2EClient, + SpendLogRow, + require_successful_call, + unique_marker, +) + +pytestmark = pytest.mark.e2e + + +def _f(value: object) -> float: + return float(value) if value is not None else 0.0 # type: ignore[arg-type] + + +def _i(value: object) -> int: + return int(value) if value is not None else 0 # type: ignore[arg-type] + + +def _s(value: object) -> str: + return str(value) if value is not None else "" + + +def _summarize(rows: List[SpendLogRow]) -> List[Dict[str, object]]: + keys = ( + "request_id", + "model", + "spend", + "status", + "cache_hit", + "prompt_tokens", + "completion_tokens", + "total_tokens", + ) + return [{k: r.get(k) for k in keys} for r in rows] + + +def _require_row( + rows: List[SpendLogRow], predicate: Callable[[SpendLogRow], bool], what: str +) -> SpendLogRow: + matches = [r for r in rows if predicate(r)] + assert matches, ( + f"no SpendLogs row {what} after polling; saw {len(rows)} row(s): " + f"{_summarize(rows)}" + ) + return matches[0] + + +def test_chat_completion_writes_nonzero_spend_row( + client: SpendE2EClient, scoped_key: str +) -> None: + result = client.chat( + scoped_key, + "gemini-2.5-flash", + f"reply with one word {unique_marker()}", + extra_body={"max_tokens": 16}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(_s(r.get("status")) == "success" for r in rs), + ) + row = _require_row( + rows, lambda r: _s(r.get("status")) == "success", "for the chat call" + ) + + assert _f(row.get("spend")) > 0, f"chat row should cost > 0: {_summarize(rows)}" + assert _s(row.get("status")) == "success" + assert _s(row.get("cache_hit")) != "True", "fresh call must not be a cache hit" + assert "gemini-2.5-flash" in _s(row.get("model")) + + prompt = _i(row.get("prompt_tokens")) + completion = _i(row.get("completion_tokens")) + total = _i(row.get("total_tokens")) + assert prompt > 0 and completion > 0 + assert total == prompt + completion, f"token arithmetic broken: {_summarize(rows)}" + + if result.response_id: + assert any( + _s(r.get("request_id")) == result.response_id for r in rows + ), f"row request_id != client response.id ({result.response_id})" + + +def test_streaming_chat_completion_tracks_spend( + client: SpendE2EClient, scoped_key: str +) -> None: + result = client.chat( + scoped_key, + "gemini-2.5-flash", + f"count to three {unique_marker()}", + stream=True, + extra_body={"max_tokens": 64}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs), + ) + row = _require_row( + rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the stream" + ) + prompt = _i(row.get("prompt_tokens")) + completion = _i(row.get("completion_tokens")) + assert prompt > 0 and completion > 0, f"streaming tokens not tracked: {_summarize(rows)}" + assert _i(row.get("total_tokens")) == prompt + completion + + +def test_embedding_writes_nonzero_spend_row( + client: SpendE2EClient, scoped_key: str +) -> None: + result = client.embed( + scoped_key, + "openai-text-embedding-3-small", + f"vectorize this sentence {unique_marker()}", + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs) + ) + row = _require_row( + rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the embedding" + ) + assert _i(row.get("prompt_tokens")) > 0 + assert _i(row.get("completion_tokens")) == 0, "embeddings have no completion tokens" + assert "text-embedding-3-small" in _s(row.get("model")) + + +def test_cache_hit_is_zero_cost_and_suffixed( + client: SpendE2EClient, scoped_key: str +) -> None: + # Unique marker shared by both calls: call 1 is a guaranteed cache MISS (fresh + # content, paid), call 2 repeats the identical request and HITS the cache just + # populated. The marker keeps each run isolated - a fixed prompt would persist + # in the shared response cache across runs and make both calls hit (flaky). + prompt = f"What is the capital of France? Answer in one word. {unique_marker()}" + first = client.chat( + scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16} + ) + require_successful_call(first) + second = client.chat( + scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16} + ) + require_successful_call(second) + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(_s(r.get("cache_hit")) == "True" for r in rs), + ) + cache_rows = [r for r in rows if _s(r.get("cache_hit")) == "True"] + if not cache_rows: + pytest.skip( + "no cache-hit row observed; caching may be disabled on this proxy. " + f"rows seen: {_summarize(rows)}" + ) + + cache_row = cache_rows[0] + assert _f(cache_row.get("spend")) == 0.0, ( + f"cache hit was charged (double-charge regression): {_summarize(rows)}" + ) + assert "_cache_hit" in _s(cache_row.get("request_id")), ( + "cache-hit row missing the _cache_hit request_id suffix; " + "duplicate-key collisions will silently drop rows" + ) + paid_rows = [r for r in rows if _s(r.get("cache_hit")) != "True"] + assert any(_f(r.get("spend")) > 0 for r in paid_rows), ( + f"the non-cached call should still be charged: {_summarize(rows)}" + ) + + +def test_key_spend_equals_sum_of_logs( + client: SpendE2EClient, scoped_key: str +) -> None: + for _ in range(2): + result = client.chat( + scoped_key, + "gemini-2.5-flash", + f"say hi {unique_marker()}", + extra_body={"max_tokens": 16}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + min_rows=2, + predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0, + ) + assert len(rows) >= 2, f"expected >=2 rows for the key, saw {_summarize(rows)}" + logs_total = sum(_f(r.get("spend")) for r in rows) + assert logs_total > 0 + + key_spend = client.poll_key_spend(scoped_key, minimum=logs_total * 0.999) + assert key_spend == pytest.approx(logs_total, rel=1e-2, abs=1e-9), ( + f"key aggregate {key_spend} != sum of logs {logs_total}; " + f"rows: {_summarize(rows)}" + ) + + +def test_request_tags_round_trip( + client: SpendE2EClient, scoped_key: str +) -> None: + tag = f"e2e-spend-{unique_marker()}" + result = client.chat( + scoped_key, + "gemini-2.5-flash", + "tagged request", + metadata={"tags": [tag]}, + extra_body={"max_tokens": 16}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(tag in _s(r.get("request_tags")) for r in rs), + ) + _require_row( + rows, + lambda r: tag in _s(r.get("request_tags")), + f"carrying request tag {tag!r}", + ) + + +def test_tag_spend_matches_sum_of_tagged_logs( + client: SpendE2EClient, scoped_key: str +) -> None: + # Unique tag so /spend/tags can't be polluted by other rows; unique content + # per call so both are fresh misses (paid), not cache hits. + tag = f"e2e-tagspend-{unique_marker()}" + for _ in range(2): + result = client.chat( + scoped_key, + "gemini-2.5-flash", + f"hi {unique_marker()}", + metadata={"tags": [tag]}, + extra_body={"max_tokens": 16}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + min_rows=2, + predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0, + ) + tagged = [r for r in rows if tag in _s(r.get("request_tags"))] + assert len(tagged) >= 2, f"expected 2 tagged rows, saw {_summarize(rows)}" + logs_total = sum(_f(r.get("spend")) for r in tagged) + assert logs_total > 0 + + entry = client.poll_tag_spend(tag, minimum=logs_total * 0.999) + assert entry is not None, f"tag {tag!r} never appeared in /spend/tags" + assert _f(entry.get("total_spend")) == pytest.approx( + logs_total, rel=1e-2, abs=1e-9 + ), f"/spend/tags total_spend {entry} != sum of tagged rows {logs_total}" + assert _i(entry.get("log_count")) == len(tagged), ( + f"/spend/tags log_count {entry.get('log_count')} != tagged rows {len(tagged)}" + ) + + +def test_end_user_spend_attributed_on_row( + client: SpendE2EClient, scoped_key: str, resources: ResourceManager +) -> None: + customer = resources.customer(f"e2e-cust-{unique_marker()}") + result = client.chat( + scoped_key, + "gemini-2.5-flash", + "hi", + extra_body={"user": customer, "max_tokens": 16}, + ) + require_successful_call(result) + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(_s(r.get("end_user")) == customer for r in rs), + ) + row = _require_row( + rows, + lambda r: _s(r.get("end_user")) == customer, + f"attributed to end_user {customer!r}", + ) + assert _f(row.get("spend")) > 0, f"end-user row should cost > 0: {_summarize(rows)}" + + +def test_failure_call_writes_failure_status_row( + client: SpendE2EClient, scoped_key: str +) -> None: + result = client.chat( + scoped_key, "gemini-2.5-flash", "", extra_body={"max_tokens": 1} + ) + if result.ok: + pytest.skip("call unexpectedly succeeded; could not induce a failure row") + + rows = client.poll_logs_for_key( + scoped_key, + predicate=lambda rs: any(_s(r.get("status")) == "failure" for r in rs), + ) + failure_rows = [r for r in rows if _s(r.get("status")) == "failure"] + if not failure_rows: + pytest.skip( + "no failure-status row was logged for the rejected call; " + "failure logging is environment-specific" + ) + assert _f(failure_rows[0].get("spend")) == 0.0, "failed call must not be charged" + + +def test_spend_calculate_returns_nonzero_cost(client: SpendE2EClient) -> None: + cost = client.calculate_spend( + "gemini-2.5-flash", "estimate the cost of this request" + ) + assert cost > 0, ( + "/spend/calculate returned 0 for gemini-2.5-flash; " + "cost map may be missing this model" + )