mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
remove e2e_tests folder
This commit is contained in:
parent
2c24bd6366
commit
8786e301bf
25 changed files with 0 additions and 2355 deletions
|
|
@ -1,93 +0,0 @@
|
|||
# Budget Code Matrix
|
||||
|
||||
What LiteLLM actually implements for budgets: every entity that can carry a dollar
|
||||
budget, how the limit is enforced, and where in the code it happens. This is the
|
||||
"what we support" reference; the companion `BUDGET_TEST_COVERAGE_MATRIX.md` maps
|
||||
each row to its tests and the e2e gaps.
|
||||
|
||||
Over-budget surfaces as a `budget_exceeded` error (the live suite
|
||||
`tests/otel_tests/test_e2e_budgeting.py` asserts `type == "budget_exceeded"`,
|
||||
`code == "429"`); the underlying `BudgetExceededError` is defined in
|
||||
`litellm/exceptions.py` (`status_code=400`). Enforcement runs in `common_checks()`
|
||||
/ `auth_checks.py` at auth time, plus pre-call reservation in
|
||||
`budget_reservation.py`.
|
||||
|
||||
Legend for "Enforced": **block** = request rejected; **filter** = router skips the
|
||||
deployment; **alert** = notify only, request proceeds.
|
||||
|
||||
---
|
||||
|
||||
## 1. Per-entity dollar budgets
|
||||
|
||||
| Entity | Budget stored | Hard `max_budget` | Soft budget | Per-window | Model budget | Reset by `budget_duration` |
|
||||
|--------|---------------|-------------------|-------------|------------|--------------|----------------------------|
|
||||
| API key | `LiteLLM_VerificationToken` (direct cols + `budget_id` FK) | block (`_virtual_key_max_budget_check`) | alert (`_virtual_key_soft_budget_check`) + 80% alert | block (`_virtual_key_multi_budget_check`) | block (`model_max_budget_limiter.is_key_within_model_budget`) | keys reset job |
|
||||
| Internal user | `LiteLLM_UserTable` (direct cols) | block (`common_checks`, only when not on a team) | - | - | via `model_max_budget` json | users reset job |
|
||||
| Team | `LiteLLM_TeamTable` (direct cols) | block (`_team_max_budget_check`) | alert (`_team_soft_budget_check`) | block (`_team_multi_budget_check`) | via `model_max_budget` | teams reset job |
|
||||
| Team member | `LiteLLM_TeamMembership` -> `LiteLLM_BudgetTable` | block (`_check_team_member_budget`) | - | - | - | budget-table reset job |
|
||||
| End-user / customer | `LiteLLM_EndUserTable` -> `LiteLLM_BudgetTable` | block (`_check_end_user_budget`) | - | - | block (`is_end_user_within_model_budget`) | budget-table reset job |
|
||||
| Organization | `LiteLLM_OrganizationTable` -> `LiteLLM_BudgetTable` | block (`_organization_max_budget_check`) | - | - | via budget-table | budget-table reset job |
|
||||
| Tag | `LiteLLM_TagTable` -> `LiteLLM_BudgetTable` | block (`_tag_max_budget_check`) | - | - | via budget-table | budget-table reset job |
|
||||
| Project | `LiteLLM_ProjectTable` -> `LiteLLM_BudgetTable` | block (`_project_max_budget_check`) | alert (`_project_soft_budget_check`) | - | - | budget-table reset job |
|
||||
| Provider (router) | config `provider_budget_config` (in-memory) | filter (`router_strategy/budget_limiter`) | - | yes (time window) | - | window TTL |
|
||||
| Global proxy | `litellm.max_budget` (config) | block (`_global_proxy_budget_check`) | - | - | - | - |
|
||||
|
||||
Notes / flags from the code:
|
||||
- **User budget only enforced off-team**: `common_checks` skips the personal-user
|
||||
budget when the key belongs to a team (team budget governs instead).
|
||||
- **Comparison operators are inconsistent**: key/user use `>=`, team/end-user main
|
||||
budget use `>`. Spend exactly at `max_budget` blocks a key but not a team.
|
||||
- **Provider budgets are filter-only**: an over-budget provider is removed from
|
||||
routing; if all are over budget the router raises
|
||||
`no_deployments_with_provider_budget_routing` (not a per-entity block).
|
||||
- **Enforcement timing differs by entity**: key / user / org / team-member / tag /
|
||||
model enforce off real-time reservation counters (block within ~2 calls);
|
||||
**end-user** enforcement reads `EndUserTable.spend`, which only updates on the
|
||||
`proxy_batch_write_at` flush, so it lags by that interval (verified live).
|
||||
|
||||
## 2. Budget mechanisms
|
||||
|
||||
| Mechanism | What it does | Code |
|
||||
|-----------|--------------|------|
|
||||
| Pre-call reservation | Estimates max request cost, atomically reserves against redis spend counters for key/team/user/end_user/tag/team_member/org before the call; blocks if a counter would exceed | `spend_tracking/budget_reservation.py` |
|
||||
| Post-call reconciliation | Adjusts the reservation to the actual cost once known | `reconcile_budget_reservation` |
|
||||
| Read-time enforcement | Auth-time check of current spend vs `max_budget` | `auth_checks.common_checks` + per-entity `_*_max_budget_check` |
|
||||
| Soft budget / alerts | At `soft_budget` (or 80% of max) fire Slack/email alert, do not block | `_virtual_key_soft_budget_check`, `_team_soft_budget_check`, `budget_alerts` |
|
||||
| Multi-window budgets | `budget_limits` list of `{budget_duration, max_budget}`; each window enforced + reset independently | `_virtual_key_multi_budget_check`, `reset_budget_windows` |
|
||||
| Model-level budgets | `model_max_budget` dict (per model: `budget_limit` + `time_period`) on key/user/end_user | `hooks/model_max_budget_limiter.py` |
|
||||
| Reset by duration | Job zeros `spend`, recomputes `budget_reset_at = now + duration_in_seconds(budget_duration)`, invalidates redis counters | `common_utils/reset_budget_job.py`, `duration_parser.duration_in_seconds` |
|
||||
| Zero-cost bypass | Models with no configured price bypass budget reservation | `budget_reservation` zero-cost path |
|
||||
|
||||
## 3. Budget management surface (endpoints)
|
||||
|
||||
| Action | Endpoint | Handler |
|
||||
|--------|----------|---------|
|
||||
| Create budget | `POST /budget/new` | `new_budget` |
|
||||
| Update budget | `POST /budget/update` | `update_budget` |
|
||||
| Budget info | `POST /budget/info` (`{"budgets": [id]}`) | `info_budget` |
|
||||
| Budget settings | `GET /budget/settings` | `budget_settings` |
|
||||
| List budgets | `GET /budget/list` | `list_budget` |
|
||||
| Delete budget | `POST /budget/delete` (`{"id": id}`) | `delete_budget` |
|
||||
| Set on key | `POST /key/generate`, `/key/update` (`max_budget`, `soft_budget`, `budget_duration`, `model_max_budget`, `budget_id`) | key mgmt |
|
||||
| Set on user | `POST /user/new` (`max_budget`, `budget_duration`) | internal user |
|
||||
| Set on team | `POST /team/new` (`max_budget`, `soft_budget`, `team_member_budget`) | team |
|
||||
| Set on team member | `POST /team/member_add` (`max_budget_in_team`) | team |
|
||||
| Set on org | `POST /organization/new` (`max_budget`, `soft_budget`, `model_max_budget`) | org |
|
||||
| Set on customer | `POST /customer/new`, `/customer/update` (`max_budget`, `budget_id`) | customer |
|
||||
| Set on tag | `POST /tag/new`, `/tag/update` (`max_budget`) | tag mgmt |
|
||||
| Read budget+spend | `/key/info`, `/user/info`, `/team/info`, `/organization/info`, `/customer/info`, `/budget/info` | per-entity info |
|
||||
|
||||
Endpoint method/shape gotchas verified live: `/organization/delete` is **DELETE**
|
||||
with `{"organization_ids": [id]}`; `/budget/info` takes `{"budgets": [id]}`;
|
||||
`model_max_budget` entries use `{"budget_limit", "time_period"}`.
|
||||
|
||||
## 4. Config knobs
|
||||
|
||||
| Setting | Effect |
|
||||
|---------|--------|
|
||||
| `litellm.max_budget` | proxy-wide hard cap (global proxy budget) |
|
||||
| `max_internal_user_budget` / `default_max_internal_user_budget` | default `max_budget` for internal users |
|
||||
| `internal_user_budget_duration` | default reset duration for internal users |
|
||||
| `max_end_user_budget` / `max_end_user_budget_id` | default budget for end-users |
|
||||
| `default_team_params` | default `max_budget` / `budget_duration` / limits for teams |
|
||||
| `provider_budget_config` (router) | per-provider spend caps + windows |
|
||||
|
|
@ -1,78 +0,0 @@
|
|||
# Budget Test Coverage Matrix
|
||||
|
||||
Maps every row of `BUDGET_CODE_MATRIX.md` (what LiteLLM implements) to its tests
|
||||
and level, then marks the live e2e coverage this suite adds.
|
||||
|
||||
Levels: `unit` mocked (`AsyncMock` on `get_current_spend`/prisma); `router` live
|
||||
router with fake deployments; `live-e2e` real proxy, real key/team, real requests
|
||||
until blocked. Status: `covered` / `partial` / `gap`.
|
||||
|
||||
Pre-existing live coverage outside this suite:
|
||||
- `tests/otel_tests/test_e2e_budgeting.py` - key + team enforcement, budget update.
|
||||
- `tests/local_testing/test_router_budget_limiter.py` - provider / tag / deployment
|
||||
budgets at the router.
|
||||
|
||||
This suite (`tests/e2e_tests/budgets/`) adds the missing live coverage and runs
|
||||
on the shared lifecycle (every entity it creates is deleted on teardown).
|
||||
|
||||
---
|
||||
|
||||
## Per-entity enforcement
|
||||
|
||||
| Entity | Unit | Pre-existing live | This suite (live) | Status |
|
||||
|--------|------|-------------------|-------------------|--------|
|
||||
| API key | `test_budget_reservation.py`, `test_max_budget_limiter.py` | `otel_tests` | `test_budget_enforcement_e2e::test_key_budget_blocks` | **covered** |
|
||||
| Team | `test_team_budget_limits.py` | `otel_tests` | (org test builds a team) | **covered** |
|
||||
| Internal user | auth unit tests | - | `test_internal_user_budget_blocks` | **covered (new)** |
|
||||
| Team member | `test_team_member_budget.py` | - | `test_team_member_budget_blocks` | **covered (new)** |
|
||||
| End-user / customer | `test_custom_auth_end_user_budget.py` | - | `test_end_user_budget_blocks` | **covered (new)** |
|
||||
| Organization | `test_organization_budget_enforcement.py` (flagged weak) | - | `test_organization_budget_blocks` | **covered (new)** |
|
||||
| Tag (proxy-level) | - | router only | `test_tag_budget_e2e::test_tag_budget_blocks_tagged_requests` | **covered (new)** |
|
||||
| Model-level (`model_max_budget`) | `test_unit_test_max_model_budget_limiter.py` | - | `test_model_max_budget_e2e::test_model_max_budget_isolates_per_model` | **covered (new)** |
|
||||
| Provider (router) | `test_budget_limiter_hotpath.py` | `test_router_budget_limiter.py` | - | **covered** (router) |
|
||||
| Global proxy (`litellm.max_budget`) | unit | - | - | **gap** (needs a config-level cap; not key-settable) |
|
||||
|
||||
## Budget mechanisms
|
||||
|
||||
| Mechanism | Unit | This suite (live) | Status |
|
||||
|-----------|------|-------------------|--------|
|
||||
| Pre-call reservation | `test_budget_reservation.py` | exercised by every enforcement test | **partial** |
|
||||
| Soft budget / alerts | `SlackAlerting/test_budget_alert_types.py` | `test_soft_budget_e2e::test_soft_budget_does_not_block` | **covered (new)** (block-vs-alert; the alert side-effect itself stays unit) |
|
||||
| Budget CRUD | `test_budget_endpoints.py` | `test_budget_crud_e2e` (roundtrip + delete) | **covered (new)** |
|
||||
| Reset scheduling | `test_proxy_budget_reset.py` | `test_budget_crud_e2e::test_budget_duration_schedules_reset_on_key` | **covered (new)** (scheduling; actual zeroing is time-dependent -> unit) |
|
||||
| Multi-window budgets | `test_multi_budget_windows.py` | - | **gap** (window setup is fiddly; left to unit for now) |
|
||||
| Read budget+spend | `test_spend_management_endpoints.py` | `/key/info` asserted in CRUD + enforcement | **partial** |
|
||||
|
||||
## Remaining gaps (intentionally not live-tested)
|
||||
|
||||
- **Global proxy budget** (`litellm.max_budget`): set via proxy config, not a
|
||||
per-key API, so it needs a dedicated proxy boot with that config rather than a
|
||||
runtime-created entity. Out of scope for the per-entity suite.
|
||||
- **Multi-window budgets**: the `budget_limits` list shape and per-window reset are
|
||||
covered by `test_multi_budget_windows.py` (unit); a live version would need to
|
||||
wait out a short window to see the reset, which is time-dependent.
|
||||
- **Soft-budget alert delivery**: whether the Slack/email actually fires is not
|
||||
observable from the proxy API; unit tests own that. The live test pins the
|
||||
load-bearing behavior (soft does not block).
|
||||
- **Reset zeroing after the window elapses**: time-dependent; unit tests own the
|
||||
reset-job logic. The live test pins that `budget_reset_at` is scheduled.
|
||||
|
||||
## This suite's files
|
||||
|
||||
| File | Covers |
|
||||
|------|--------|
|
||||
| `test_budget_enforcement_e2e.py` | key / internal-user / end-user / organization / team-member hard enforcement |
|
||||
| `test_model_max_budget_e2e.py` | per-model caps isolate by model |
|
||||
| `test_soft_budget_e2e.py` | soft budget alerts but does not block |
|
||||
| `test_tag_budget_e2e.py` | proxy-level tag budget blocks tagged requests, spares others |
|
||||
| `test_budget_crud_e2e.py` | `/budget/*` CRUD roundtrip + delete + `budget_reset_at` scheduling |
|
||||
|
||||
## Pattern + timing
|
||||
|
||||
Create the entity with a tiny `max_budget`, drive spend until a `budget_exceeded`
|
||||
block. The enforcement helper is two-phase: a fast warmup (key/user/org/member/tag/
|
||||
model block within ~2 calls off real-time counters), then a poll across the ~60s
|
||||
batch-write window (end-user enforcement reads table spend that lags). Skip on a
|
||||
non-budget error (provider down / key missing); fail if the budget is never
|
||||
enforced. Chat tests use `gpt-5.5` (the model with a working key on the reference
|
||||
proxy); swap the literal if your proxy differs.
|
||||
|
|
@ -1,185 +0,0 @@
|
|||
"""Client for budget e2e tests: the shared ProxyClient plus budget-bearing entity
|
||||
management (user / team / team-member / org / customer / tag / budget-table) and
|
||||
info reads.
|
||||
|
||||
Over-budget surfaces as a ``budget_exceeded`` error; ``is_budget_block`` detects it
|
||||
on a CallResult. Create methods return the new id and raise on failure; tests
|
||||
register the matching delete with ``resources.defer(...)`` for cleanup.
|
||||
"""
|
||||
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from proxy_client import CallResult, ProxyClient, _auth, proxy_client_kwargs
|
||||
|
||||
|
||||
def is_budget_block(result: CallResult) -> bool:
|
||||
"""True if the call was rejected for being over budget (vs a provider error)."""
|
||||
return not result.ok and "budget_exceeded" in result.body
|
||||
|
||||
|
||||
def model_budget(model: str, limit: float, period: str = "30d") -> dict:
|
||||
"""A model_max_budget dict entry: per-model cap with a reset window."""
|
||||
return {model: {"budget_limit": limit, "time_period": period}}
|
||||
|
||||
|
||||
class BudgetClient(ProxyClient):
|
||||
def _post(self, path: str, body: Dict[str, object]) -> Dict[str, object]:
|
||||
resp = requests.post(
|
||||
f"{self._base_url}{path}",
|
||||
headers=_auth(self._master_key),
|
||||
json=body,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return dict(resp.json()) if resp.text else {}
|
||||
|
||||
def _get(self, path: str, params: Dict[str, str]) -> Dict[str, object]:
|
||||
resp = requests.get(
|
||||
f"{self._base_url}{path}",
|
||||
headers=_auth(self._master_key),
|
||||
params=params,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return dict(resp.json())
|
||||
|
||||
def _delete(self, path: str, body: Dict[str, object]) -> None:
|
||||
try:
|
||||
requests.delete(
|
||||
f"{self._base_url}{path}",
|
||||
headers=_auth(self._master_key),
|
||||
json=body,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
except requests.RequestException:
|
||||
pass
|
||||
|
||||
def _post_quiet(self, path: str, body: Dict[str, object]) -> None:
|
||||
try:
|
||||
requests.post(
|
||||
f"{self._base_url}{path}",
|
||||
headers=_auth(self._master_key),
|
||||
json=body,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
except requests.RequestException:
|
||||
pass
|
||||
|
||||
# ---- internal user --------------------------------------------------
|
||||
|
||||
def create_user(
|
||||
self, *, max_budget: float, budget_duration: Optional[str] = None
|
||||
) -> str:
|
||||
body: Dict[str, object] = {"max_budget": max_budget}
|
||||
if budget_duration is not None:
|
||||
body["budget_duration"] = budget_duration
|
||||
return str(self._post("/user/new", body)["user_id"])
|
||||
|
||||
def delete_user(self, user_id: str) -> None:
|
||||
self._post_quiet("/user/delete", {"user_ids": [user_id]})
|
||||
|
||||
def user_info(self, user_id: str) -> Dict[str, object]:
|
||||
return self._get("/user/info", {"user_id": user_id})
|
||||
|
||||
# ---- customer / end-user -------------------------------------------
|
||||
|
||||
def create_customer(self, customer_id: str, *, max_budget: float) -> str:
|
||||
self._post("/customer/new", {"user_id": customer_id, "max_budget": max_budget})
|
||||
return customer_id
|
||||
|
||||
def customer_info(self, customer_id: str) -> Dict[str, object]:
|
||||
return self._get("/customer/info", {"end_user_id": customer_id})
|
||||
|
||||
# ---- organization ---------------------------------------------------
|
||||
|
||||
def create_org(self, *, max_budget: float, alias: str) -> str:
|
||||
return str(
|
||||
self._post(
|
||||
"/organization/new",
|
||||
{"organization_alias": alias, "max_budget": max_budget},
|
||||
)["organization_id"]
|
||||
)
|
||||
|
||||
def delete_org(self, org_id: str) -> None:
|
||||
self._delete("/organization/delete", {"organization_ids": [org_id]})
|
||||
|
||||
# ---- team -----------------------------------------------------------
|
||||
|
||||
def create_team(
|
||||
self,
|
||||
*,
|
||||
alias: str,
|
||||
max_budget: Optional[float] = None,
|
||||
organization_id: Optional[str] = None,
|
||||
extra: Optional[Dict[str, object]] = None,
|
||||
) -> str:
|
||||
body: Dict[str, object] = {"team_alias": alias}
|
||||
if max_budget is not None:
|
||||
body["max_budget"] = max_budget
|
||||
if organization_id is not None:
|
||||
body["organization_id"] = organization_id
|
||||
if extra:
|
||||
body.update(extra)
|
||||
return str(self._post("/team/new", body)["team_id"])
|
||||
|
||||
def delete_team(self, team_id: str) -> None:
|
||||
self._post_quiet("/team/delete", {"team_ids": [team_id]})
|
||||
|
||||
def team_info(self, team_id: str) -> Dict[str, object]:
|
||||
return self._get("/team/info", {"team_id": team_id})
|
||||
|
||||
def add_team_member(
|
||||
self, team_id: str, user_id: str, *, max_budget_in_team: Optional[float] = None
|
||||
) -> None:
|
||||
body: Dict[str, object] = {
|
||||
"team_id": team_id,
|
||||
"member": {"role": "user", "user_id": user_id},
|
||||
}
|
||||
if max_budget_in_team is not None:
|
||||
body["max_budget_in_team"] = max_budget_in_team
|
||||
self._post("/team/member_add", body)
|
||||
|
||||
# ---- tag ------------------------------------------------------------
|
||||
|
||||
def create_tag(self, name: str, *, max_budget: float) -> str:
|
||||
self._post("/tag/new", {"name": name, "max_budget": max_budget})
|
||||
return name
|
||||
|
||||
def delete_tag(self, name: str) -> None:
|
||||
self._post_quiet("/tag/delete", {"name": name})
|
||||
|
||||
# ---- budget table ---------------------------------------------------
|
||||
|
||||
def create_budget(
|
||||
self,
|
||||
*,
|
||||
max_budget: float,
|
||||
soft_budget: Optional[float] = None,
|
||||
budget_duration: Optional[str] = None,
|
||||
) -> str:
|
||||
body: Dict[str, object] = {"max_budget": max_budget}
|
||||
if soft_budget is not None:
|
||||
body["soft_budget"] = soft_budget
|
||||
if budget_duration is not None:
|
||||
body["budget_duration"] = budget_duration
|
||||
return str(self._post("/budget/new", body)["budget_id"])
|
||||
|
||||
def delete_budget(self, budget_id: str) -> None:
|
||||
self._post_quiet("/budget/delete", {"id": budget_id})
|
||||
|
||||
def budget_info(self, budget_id: str) -> List[Dict[str, object]]:
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/budget/info",
|
||||
headers=_auth(self._master_key),
|
||||
json={"budgets": [budget_id]},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return [dict(row) for row in data] if isinstance(data, list) else []
|
||||
|
||||
|
||||
def build_client() -> BudgetClient:
|
||||
return BudgetClient(**proxy_client_kwargs())
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
"""Budgets suite's `client` fixture.
|
||||
|
||||
The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
|
||||
live in the parent tests/e2e_tests/conftest.py. BudgetClient subclasses
|
||||
ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources`
|
||||
fixture cleans up keys; tests register entity deletes via `resources.defer(...)`.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, build_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> BudgetClient:
|
||||
return build_client()
|
||||
|
|
@ -1,68 +0,0 @@
|
|||
"""Live e2e for the budget management surface (no LLM calls, fast).
|
||||
|
||||
Covers the budget-table CRUD round-trip and that `budget_duration` schedules a
|
||||
`budget_reset_at`. The actual zeroing after the window is time-dependent, so we
|
||||
assert the reset is *scheduled* (now + duration), not waited out.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient
|
||||
from lifecycle import ResourceManager
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
def _f(value: object) -> float:
|
||||
return float(value) if value is not None else 0.0 # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_budget_crud_roundtrip(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
budget_id = client.create_budget(
|
||||
max_budget=12.5, soft_budget=10.0, budget_duration="30d"
|
||||
)
|
||||
resources.defer(lambda: client.delete_budget(budget_id))
|
||||
|
||||
rows = client.budget_info(budget_id)
|
||||
assert rows, f"/budget/info returned nothing for {budget_id}"
|
||||
row = rows[0]
|
||||
assert _f(row.get("max_budget")) == 12.5
|
||||
assert _f(row.get("soft_budget")) == 10.0
|
||||
assert row.get("budget_reset_at"), "budget_duration did not schedule a reset"
|
||||
|
||||
# Attach the budget to a key and confirm the key reflects it.
|
||||
key = client.generate_key(extra_params={"budget_id": budget_id})
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
info = client.key_info(key)
|
||||
linked = info.get("litellm_budget_table") or {}
|
||||
assert (
|
||||
info.get("budget_id") == budget_id or _f(linked.get("max_budget")) == 12.5
|
||||
), f"key does not reflect attached budget: {info.get('budget_id')}, {linked}"
|
||||
|
||||
|
||||
def test_budget_delete_removes_it(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
budget_id = client.create_budget(max_budget=1.0)
|
||||
client.delete_budget(budget_id)
|
||||
assert client.budget_info(budget_id) == [], "budget still present after delete"
|
||||
|
||||
|
||||
def test_budget_duration_schedules_reset_on_key(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = client.generate_key(
|
||||
max_budget=10.0, extra_params={"budget_duration": "30d"}
|
||||
)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
reset_at = client.key_info(key).get("budget_reset_at")
|
||||
assert reset_at, "budget_duration did not set budget_reset_at on the key"
|
||||
|
||||
reset_dt = datetime.fromisoformat(str(reset_at).replace("Z", "+00:00"))
|
||||
days_out = (reset_dt - datetime.now(timezone.utc)).total_seconds() / 86400
|
||||
assert 28 < days_out < 32, f"reset ~30d expected, got {days_out:.1f}d out"
|
||||
|
|
@ -1,119 +0,0 @@
|
|||
"""Live e2e: a tiny max_budget on an entity actually blocks requests.
|
||||
|
||||
Mirrors tests/otel_tests/test_e2e_budgeting.py (call until `budget_exceeded`) but
|
||||
covers the entities that had NO live coverage - internal user, end-user,
|
||||
organization, team member - and runs in this suite so the shared `resources`
|
||||
teardown deletes every entity created. See BUDGET_TEST_COVERAGE_MATRIX.md.
|
||||
|
||||
Skip on environment, fail on behavior: a non-budget error (provider down) skips;
|
||||
if calls never get blocked, budget enforcement is broken -> fail.
|
||||
"""
|
||||
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block
|
||||
from lifecycle import ResourceManager
|
||||
from proxy_client import require_successful_call, unique_marker
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
TINY_BUDGET = 1e-6 # one real call's spend exceeds this, so the block lands fast
|
||||
|
||||
|
||||
def _assert_budget_blocks(
|
||||
client: BudgetClient, key: str, model: str, *, user: str = ""
|
||||
) -> None:
|
||||
"""Drive spend over the entity's budget and assert a call is blocked.
|
||||
|
||||
Two-phase so it's robust across entities: key/user/org enforce off real-time
|
||||
reservation counters (block within a couple calls), but end-user enforcement
|
||||
reads the table spend that only updates on the ~60s batch write - so after a
|
||||
warmup we poll for the block across that window. Skip on a non-budget error;
|
||||
fail if the budget is never enforced.
|
||||
"""
|
||||
extra: dict = {"max_tokens": 16}
|
||||
if user:
|
||||
extra["user"] = user
|
||||
|
||||
def _call(label: str):
|
||||
result = client.chat(key, model, f"{label} {unique_marker()}", extra_body=extra)
|
||||
if not result.ok and not is_budget_block(result):
|
||||
require_successful_call(result) # non-budget error -> skip
|
||||
return result
|
||||
|
||||
for _ in range(2): # warmup: incur spend (fast entities block here)
|
||||
if is_budget_block(_call("warmup")):
|
||||
return
|
||||
time.sleep(0.5)
|
||||
|
||||
deadline = time.monotonic() + 100 # cover the batch-write propagation window
|
||||
while time.monotonic() < deadline:
|
||||
if is_budget_block(_call("probe")):
|
||||
return
|
||||
time.sleep(5)
|
||||
pytest.fail("budget not enforced within 100s")
|
||||
|
||||
|
||||
def test_key_budget_blocks(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
key = client.generate_key(max_budget=TINY_BUDGET)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
_assert_budget_blocks(client, key, "gpt-5.5")
|
||||
|
||||
|
||||
def test_internal_user_budget_blocks(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
user_id = client.create_user(max_budget=TINY_BUDGET)
|
||||
resources.defer(lambda: client.delete_user(user_id))
|
||||
# personal key (no team) -> the user budget governs
|
||||
key = client.generate_key(extra_params={"user_id": user_id})
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
_assert_budget_blocks(client, key, "gpt-5.5")
|
||||
|
||||
|
||||
def test_end_user_budget_blocks(
|
||||
client: BudgetClient, scoped_key: str, resources: ResourceManager
|
||||
) -> None:
|
||||
customer = f"e2e-budget-cust-{unique_marker()}"
|
||||
client.create_customer(customer, max_budget=TINY_BUDGET)
|
||||
resources.defer(lambda: client.delete_customers([customer]))
|
||||
_assert_budget_blocks(client, scoped_key, "gpt-5.5", user=customer)
|
||||
|
||||
|
||||
def test_organization_budget_blocks(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
# Org carries the tiny budget; the team under it has none, so a block here is
|
||||
# org-level enforcement (the historically weak link).
|
||||
org_id = client.create_org(
|
||||
max_budget=TINY_BUDGET, alias=f"e2e-budget-org-{unique_marker()}"
|
||||
)
|
||||
resources.defer(lambda: client.delete_org(org_id))
|
||||
team_id = client.create_team(
|
||||
alias=f"e2e-budget-team-{unique_marker()}", organization_id=org_id
|
||||
)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
key = client.generate_key(extra_params={"team_id": team_id})
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
_assert_budget_blocks(client, key, "gpt-5.5")
|
||||
|
||||
|
||||
def test_team_member_budget_blocks(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
# Member's per-team budget is tiny while the team itself has a large budget,
|
||||
# so a block proves member-level (not team-level) enforcement.
|
||||
team_id = client.create_team(
|
||||
alias=f"e2e-budget-team-{unique_marker()}", max_budget=100.0
|
||||
)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
user_id = client.create_user(max_budget=100.0)
|
||||
resources.defer(lambda: client.delete_user(user_id))
|
||||
client.add_team_member(team_id, user_id, max_budget_in_team=TINY_BUDGET)
|
||||
key = client.generate_key(
|
||||
extra_params={"team_id": team_id, "user_id": user_id}
|
||||
)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
_assert_budget_blocks(client, key, "gpt-5.5")
|
||||
|
|
@ -1,60 +0,0 @@
|
|||
"""Live e2e: per-model budgets (`model_max_budget`) isolate by model.
|
||||
|
||||
A key caps one model tiny and leaves another generous. Exhausting the capped
|
||||
model must block *that* model while the other still works - proving the per-model
|
||||
cap is enforced independently, not as a key-wide budget. Closes the
|
||||
model_max_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md.
|
||||
"""
|
||||
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block, model_budget
|
||||
from lifecycle import ResourceManager
|
||||
from proxy_client import require_successful_call, unique_marker
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CAPPED_MODEL = "gpt-5.5"
|
||||
FREE_MODEL = "claude-haiku-4-5"
|
||||
|
||||
|
||||
def _call(client: BudgetClient, key: str, model: str):
|
||||
result = client.chat(
|
||||
key, model, f"hi {unique_marker()}", extra_body={"max_tokens": 16}
|
||||
)
|
||||
if not result.ok and not is_budget_block(result):
|
||||
require_successful_call(result) # non-budget error -> skip
|
||||
return result
|
||||
|
||||
|
||||
def test_model_max_budget_isolates_per_model(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = client.generate_key(
|
||||
extra_params={
|
||||
"model_max_budget": {
|
||||
**model_budget(CAPPED_MODEL, 1e-6),
|
||||
**model_budget(FREE_MODEL, 1000.0),
|
||||
}
|
||||
}
|
||||
)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
# Exhaust the capped model.
|
||||
blocked = False
|
||||
deadline = time.monotonic() + 60
|
||||
while time.monotonic() < deadline:
|
||||
if is_budget_block(_call(client, key, CAPPED_MODEL)):
|
||||
blocked = True
|
||||
break
|
||||
time.sleep(1)
|
||||
assert blocked, f"{CAPPED_MODEL} per-model budget never enforced"
|
||||
|
||||
# The other model shares the key but has its own (large) cap -> still works.
|
||||
other = _call(client, key, FREE_MODEL)
|
||||
require_successful_call(other) # skips if claude key absent; never a budget block
|
||||
assert not is_budget_block(other), (
|
||||
f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated"
|
||||
)
|
||||
|
|
@ -1,36 +0,0 @@
|
|||
"""Live e2e: soft_budget alerts but does NOT block.
|
||||
|
||||
A key with a tiny `soft_budget` well under a large `max_budget`: spend crosses the
|
||||
soft threshold within a couple calls, but requests keep succeeding (soft budget is
|
||||
advisory). Closes the soft_budget gap in BUDGET_TEST_COVERAGE_MATRIX.md. The alert
|
||||
side-effect (Slack/email) is not observable from the proxy API, so we assert the
|
||||
load-bearing behavior: soft != block.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block
|
||||
from lifecycle import ResourceManager
|
||||
from proxy_client import require_successful_call, unique_marker
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
def test_soft_budget_does_not_block(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
# soft far below max: spend crosses soft immediately, stays under max.
|
||||
key = client.generate_key(
|
||||
max_budget=1000.0, extra_params={"soft_budget": 1e-9}
|
||||
)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
for _ in range(3):
|
||||
result = client.chat(
|
||||
key, "gpt-5.5", f"hi {unique_marker()}", extra_body={"max_tokens": 16}
|
||||
)
|
||||
require_successful_call(result) # skip if provider unavailable
|
||||
assert not is_budget_block(result), (
|
||||
"soft_budget blocked a request; it must alert only, not block "
|
||||
f"(body={result.body[:200]})"
|
||||
)
|
||||
|
|
@ -1,58 +0,0 @@
|
|||
"""Live e2e: proxy-level tag budgets block tagged requests.
|
||||
|
||||
A tag with a tiny budget: requests carrying that tag get blocked once the tag's
|
||||
spend is exceeded, while a request with a different tag (no budget) still works.
|
||||
Closes the proxy-level tag-budget gap in BUDGET_TEST_COVERAGE_MATRIX.md (today
|
||||
only router-level tag budgets are tested).
|
||||
"""
|
||||
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block
|
||||
from lifecycle import ResourceManager
|
||||
from proxy_client import require_successful_call, unique_marker
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
TINY_BUDGET = 1e-6
|
||||
|
||||
|
||||
def _tagged_call(client: BudgetClient, key: str, tag: str):
|
||||
result = client.chat(
|
||||
key,
|
||||
"gpt-5.5",
|
||||
f"hi {unique_marker()}",
|
||||
metadata={"tags": [tag]},
|
||||
extra_body={"max_tokens": 16},
|
||||
)
|
||||
if not result.ok and not is_budget_block(result):
|
||||
require_successful_call(result) # non-budget error -> skip
|
||||
return result
|
||||
|
||||
|
||||
def test_tag_budget_blocks_tagged_requests(
|
||||
client: BudgetClient, scoped_key: str, resources: ResourceManager
|
||||
) -> None:
|
||||
budgeted_tag = f"e2e-budget-tag-{unique_marker()}"
|
||||
client.create_tag(budgeted_tag, max_budget=TINY_BUDGET)
|
||||
resources.defer(lambda: client.delete_tag(budgeted_tag))
|
||||
|
||||
# Requests under the budgeted tag get blocked once its spend is exceeded.
|
||||
blocked = False
|
||||
deadline = time.monotonic() + 60
|
||||
while time.monotonic() < deadline:
|
||||
if is_budget_block(_tagged_call(client, scoped_key, budgeted_tag)):
|
||||
blocked = True
|
||||
break
|
||||
time.sleep(1)
|
||||
assert blocked, f"tag budget for {budgeted_tag!r} never enforced"
|
||||
|
||||
# A request with an unbudgeted tag on the same key is unaffected.
|
||||
free_tag = f"e2e-free-tag-{unique_marker()}"
|
||||
other = _tagged_call(client, scoped_key, free_tag)
|
||||
require_successful_call(other)
|
||||
assert not is_budget_block(other), (
|
||||
f"unbudgeted tag {free_tag!r} was blocked by {budgeted_tag!r}'s budget"
|
||||
)
|
||||
|
|
@ -1,54 +0,0 @@
|
|||
"""Shared fixtures for all live e2e suites under tests/e2e_tests/.
|
||||
|
||||
Design rule: skip on environment, fail on behavior. If the proxy is unreachable
|
||||
the whole session skips; once a request reaches the proxy, behavior is asserted.
|
||||
|
||||
Lifecycle: the `resources` fixture maps the init -> run -> teardown contract
|
||||
(lifecycle.E2ECase) onto pytest - setup is init(), the test body is run(), and
|
||||
teardown deletes every resource the test created on the long-lived proxy.
|
||||
|
||||
Each suite provides its own `client` fixture (a lifecycle.ResourceClient); these
|
||||
shared fixtures build on it.
|
||||
"""
|
||||
|
||||
from typing import Iterator
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
|
||||
from e2e_config import PROXY_BASE_URL
|
||||
from lifecycle import ResourceClient, ResourceManager
|
||||
|
||||
|
||||
def pytest_configure(config: pytest.Config) -> None:
|
||||
config.addinivalue_line(
|
||||
"markers",
|
||||
"e2e: live test that requires a running proxy and real provider keys",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="session", autouse=True)
|
||||
def _require_live_proxy() -> None:
|
||||
"""Skip the entire session unless a proxy answers its liveness probe."""
|
||||
try:
|
||||
resp = requests.get(f"{PROXY_BASE_URL}/health/liveliness", timeout=5)
|
||||
except requests.RequestException as exc:
|
||||
pytest.skip(f"No live proxy at {PROXY_BASE_URL}: {exc}")
|
||||
return
|
||||
if resp.status_code >= 500:
|
||||
pytest.skip(f"Proxy at {PROXY_BASE_URL} returned {resp.status_code}")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def resources(client: ResourceClient) -> Iterator[ResourceManager]:
|
||||
"""init -> run -> teardown: create a manager, run the test, release resources."""
|
||||
manager = ResourceManager(client=client)
|
||||
manager.init()
|
||||
yield manager
|
||||
manager.teardown()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def scoped_key(resources: ResourceManager) -> str:
|
||||
"""A fresh all-models key per test, auto-deleted by the resources teardown."""
|
||||
return resources.key()
|
||||
|
|
@ -1,59 +0,0 @@
|
|||
# Local e2e proxy, built from the litellm_internal_staging branch.
|
||||
#
|
||||
# The image `litellm-staging:local` is built from a clean checkout of
|
||||
# litellm_internal_staging (the working tree's uv.lock has merge markers, so
|
||||
# build from a worktree, not the dirty tree):
|
||||
#
|
||||
# git worktree add /tmp/litellm-staging litellm_internal_staging
|
||||
# docker build -t litellm-staging:local /tmp/litellm-staging
|
||||
#
|
||||
# Then provide provider keys (copy .env.example -> .env, fill in keys) and run:
|
||||
#
|
||||
# docker compose -f tests/e2e_tests/docker-compose.yml up -d
|
||||
#
|
||||
# The proxy comes up on http://localhost:4000 with master key sk-1234, which is
|
||||
# what the e2e suites default to. Point the suites at it:
|
||||
#
|
||||
# LITELLM_MASTER_KEY=sk-1234 LITELLM_PROXY_URL=http://localhost:4000 \
|
||||
# .venv/bin/python -m pytest tests/e2e_tests/ -v
|
||||
|
||||
services:
|
||||
litellm:
|
||||
image: litellm-staging:local
|
||||
ports:
|
||||
- "4000:4000"
|
||||
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
||||
volumes:
|
||||
- ./docker-config.yaml:/app/config.yaml
|
||||
environment:
|
||||
DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm"
|
||||
LITELLM_MASTER_KEY: "sk-1234"
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
env_file:
|
||||
# Provider keys for the LLM-calling suites. Boots without it (those tests
|
||||
# skip); copy .env.example -> .env and `docker compose up -d` to enable.
|
||||
- path: .env
|
||||
required: false
|
||||
depends_on:
|
||||
- db
|
||||
- redis
|
||||
|
||||
db:
|
||||
image: postgres:17
|
||||
environment:
|
||||
POSTGRES_DB: litellm
|
||||
POSTGRES_USER: llmproxy
|
||||
POSTGRES_PASSWORD: dbpassword9090
|
||||
ports:
|
||||
- "5432:5432"
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
|
||||
redis:
|
||||
image: redis:7-alpine
|
||||
ports:
|
||||
- "6379:6379"
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
driver: local
|
||||
|
|
@ -1,39 +0,0 @@
|
|||
# Config for the local docker e2e proxy (see docker-compose.yml).
|
||||
# Minimal on purpose: just the models the e2e suites use + spend tracking +
|
||||
# a redis cache (for the cache-hit spend test). Provider keys come from .env.
|
||||
# Passthrough endpoints (/gemini, /anthropic) use GEMINI_API_KEY / ANTHROPIC_API_KEY
|
||||
# from the environment directly, so no model_list entry is needed for them.
|
||||
|
||||
general_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
database_url: os.environ/DATABASE_URL
|
||||
store_prompts_in_spend_logs: true
|
||||
|
||||
litellm_settings:
|
||||
drop_params: true
|
||||
cache: true
|
||||
cache_params:
|
||||
type: redis
|
||||
host: redis
|
||||
port: 6379
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-5.5
|
||||
litellm_params:
|
||||
model: openai/gpt-5.5
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: claude-haiku-4-5
|
||||
litellm_params:
|
||||
model: anthropic/claude-haiku-4-5
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
- model_name: gemini-2.5-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.5-flash
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
|
||||
- model_name: openai-text-embedding-3-small
|
||||
litellm_params:
|
||||
model: openai/text-embedding-3-small
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
"""Generic configuration for live e2e tests against a running LiteLLM proxy.
|
||||
|
||||
Shared by every e2e suite under tests/e2e_tests/. Values come from the
|
||||
environment so the same tests run against localhost or a deployed proxy.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/")
|
||||
MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234")
|
||||
|
||||
# Writes on the proxy are eventually consistent (e.g. spend rows flush on
|
||||
# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once.
|
||||
POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120"))
|
||||
POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5"))
|
||||
REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60"))
|
||||
|
|
@ -1,87 +0,0 @@
|
|||
"""Lifecycle contract and resource cleanup for stateful e2e tests.
|
||||
|
||||
Shared by every e2e suite under tests/e2e_tests/. The proxy under test is
|
||||
long-lived and never reset between tests, so anything a test creates (keys,
|
||||
customers, teams, orgs, users, guardrails, budgets, ...) persists unless
|
||||
explicitly deleted. Every check follows an init -> run -> teardown lifecycle;
|
||||
teardown releases each resource init() created, even when run() raises.
|
||||
|
||||
In pytest terms (see conftest.py): the `resources` fixture's setup is init(),
|
||||
the test body is run(), and the fixture's teardown is teardown().
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Callable, List, Protocol, runtime_checkable
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class E2ECase(Protocol):
|
||||
"""A stateful e2e check run against a long-lived proxy.
|
||||
|
||||
init() acquires resources, run() exercises behaviour and asserts, teardown()
|
||||
releases everything init() created. teardown() must run even if run() raises.
|
||||
"""
|
||||
|
||||
def init(self) -> None: ...
|
||||
|
||||
def run(self) -> None: ...
|
||||
|
||||
def teardown(self) -> None: ...
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class ResourceClient(Protocol):
|
||||
"""Proxy operations the convenience creators use. Resource types without a
|
||||
creator here are handled generically via ResourceManager.defer()."""
|
||||
|
||||
def generate_key(self, *, models: List[str]) -> str: ...
|
||||
|
||||
def delete_key(self, key: str) -> None: ...
|
||||
|
||||
def delete_customers(self, user_ids: List[str]) -> None: ...
|
||||
|
||||
|
||||
@dataclass
|
||||
class ResourceManager:
|
||||
"""Registry of teardown actions for resources a test creates on the stateful
|
||||
proxy.
|
||||
|
||||
Not limited to any resource type: register a cleanup with ``defer()`` for a
|
||||
key, customer, team, org, user, guardrail, budget, MCP server - anything with
|
||||
a delete. The two most common resources have sugar (``key``, ``customer``);
|
||||
everything else is ``resources.defer(lambda: client.delete_team(team_id))``.
|
||||
|
||||
Cleanups run LIFO (so a resource is removed before whatever it depends on) and
|
||||
best-effort (one failing cleanup never blocks the rest).
|
||||
"""
|
||||
|
||||
client: ResourceClient
|
||||
_cleanups: List[Callable[[], None]] = field(
|
||||
default_factory=list
|
||||
) # mutable-ok: append-only teardown registry
|
||||
|
||||
def init(self) -> None:
|
||||
"""No global setup needed today; present for lifecycle symmetry."""
|
||||
return None
|
||||
|
||||
def defer(self, cleanup: Callable[[], None]) -> None:
|
||||
"""Register a teardown action for any resource the test just created."""
|
||||
self._cleanups.append(cleanup)
|
||||
|
||||
def key(self) -> str:
|
||||
"""Create an all-models virtual key; delete it on teardown."""
|
||||
key = self.client.generate_key(models=[])
|
||||
self.defer(lambda: self.client.delete_key(key))
|
||||
return key
|
||||
|
||||
def customer(self, customer_id: str) -> str:
|
||||
"""Track an end-user id (from the `user` param); delete it on teardown."""
|
||||
self.defer(lambda: self.client.delete_customers([customer_id]))
|
||||
return customer_id
|
||||
|
||||
def teardown(self) -> None:
|
||||
for cleanup in reversed(self._cleanups):
|
||||
try:
|
||||
cleanup()
|
||||
except Exception:
|
||||
pass # best-effort: a failed cleanup must not block the rest
|
||||
|
|
@ -1,84 +0,0 @@
|
|||
# LLM Translation Test Coverage Matrix
|
||||
|
||||
Scope: the proxy's two translation surfaces, end to end against a live proxy.
|
||||
|
||||
1. **Passthrough** - the client speaks the provider's NATIVE API (Gemini
|
||||
`generateContent`, Anthropic `/v1/messages`); the proxy forwards it and still
|
||||
logs a costed `SpendLogs` row (`call_type="pass_through_endpoint"`). Routes:
|
||||
`/gemini`, `/anthropic`, `/vertex_ai`, `/openai`, `/bedrock`, `/cohere`,
|
||||
`/mistral`, `/vllm`.
|
||||
2. **Non-passthrough** - the client speaks OpenAI format
|
||||
(`/chat/completions`, `/embeddings`); litellm translates to/from the provider.
|
||||
|
||||
The two axes that must work in production for each: **passthrough vs
|
||||
non-passthrough** and **streaming vs non-streaming**, with **cost logged** and
|
||||
**tool calls** working in every cell.
|
||||
|
||||
Companion: live suite `test_passthrough_e2e.py` (this directory). The
|
||||
non-passthrough chat/embedding cells are exercised by `../spend_tracking/`.
|
||||
|
||||
Levels: `live` real provider + proxy + SpendLogs row; `unit` mocked.
|
||||
Status: `covered` / `partial` / `gap`.
|
||||
|
||||
---
|
||||
|
||||
## Passthrough endpoints (native provider format)
|
||||
|
||||
| Provider | Non-streaming | Streaming | Tool calls | Cost logged | Status |
|
||||
|----------|---------------|-----------|------------|-------------|--------|
|
||||
| Gemini (`/gemini/v1beta/models/{m}:generateContent` / `:streamGenerateContent`) | live | live | live | live | **covered** |
|
||||
| Anthropic (`/anthropic/v1/messages`) | live | live | live | live | **covered** |
|
||||
| Vertex AI (`/vertex_ai/...`) | - | - | - | - | gap (gcloud auth) |
|
||||
| OpenAI / Bedrock / Cohere / Mistral / VLLM | - | - | - | - | gap |
|
||||
|
||||
Each covered cell asserts: `call_type == "pass_through_endpoint"`, `spend > 0`,
|
||||
`status == "success"`, correct `custom_llm_provider`/`model`, row correlated by the
|
||||
`x-litellm-call-id` header. Gemini non-streaming also pins `request_tags`
|
||||
propagation; streaming pins `chunks > 0` then a costed row; tool tests assert the
|
||||
provider emitted a tool call (`functionCall` / `tool_use`) and it was costed.
|
||||
|
||||
Cost on passthrough is computed in the success handler by transforming the native
|
||||
response to a `ModelResponse` and calling `litellm.completion_cost()`; for
|
||||
streaming, chunks are buffered and costed after the stream ends. This is the path
|
||||
most likely to silently break and the one a mock can't prove works.
|
||||
|
||||
## Non-passthrough endpoints (OpenAI-compatible translation)
|
||||
|
||||
| Modality | Non-streaming | Streaming | Tool calls | Cost logged | Status |
|
||||
|----------|---------------|-----------|------------|-------------|--------|
|
||||
| Chat | live (spend suite) | live (spend suite) | gap | live | partial |
|
||||
| Embeddings | live (spend suite) | n/a | n/a | live | covered |
|
||||
| Responses / image / audio / rerank / realtime | - | - | - | - | gap |
|
||||
|
||||
## This suite's files
|
||||
|
||||
| Test | Cell |
|
||||
|------|------|
|
||||
| `test_gemini_passthrough_nonstreaming_logs_cost` | gemini native, non-stream, cost + tags |
|
||||
| `test_gemini_passthrough_streaming_logs_cost` | gemini native, stream, cost |
|
||||
| `test_gemini_passthrough_tool_call_logs_cost` | gemini native, tool call, cost |
|
||||
| `test_anthropic_passthrough_nonstreaming_logs_cost` | anthropic native, non-stream, cost |
|
||||
| `test_anthropic_passthrough_streaming_logs_cost` | anthropic native, stream, cost |
|
||||
| `test_anthropic_passthrough_tool_call_logs_cost` | anthropic native, tool call, cost |
|
||||
|
||||
## Gaps
|
||||
|
||||
- Vertex / OpenAI / Bedrock / Cohere passthrough (same shape; add once the
|
||||
provider credential is configured; Vertex is closest - route exists, auth stale).
|
||||
- Non-passthrough tool calls over `/chat/completions` end to end with cost.
|
||||
- Image / audio / rerank / responses / realtime translation + cost.
|
||||
- Streaming cost-injection (`include_cost_in_streaming_usage`); passthrough on
|
||||
client disconnect (partial-usage logging).
|
||||
|
||||
## Adding a provider/modality
|
||||
|
||||
Extend `PassthroughClient` with the native call (it inherits keys, cleanup, and
|
||||
SpendLogs polling from `ProxyClient`), then add a test that calls it,
|
||||
`require_successful_call(result)`, and `_costed_row(...)`.
|
||||
|
||||
## Timing
|
||||
|
||||
Passthrough spend is logged asynchronously after the response and lands on the
|
||||
`proxy_batch_write_at` (~60s) cycle, so cost assertions poll
|
||||
`/spend/logs?request_id=<x-litellm-call-id>` to a deadline. Streaming cost is only
|
||||
known after the stream is fully consumed.
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
"""LLM-translation suite's `client` fixture.
|
||||
|
||||
The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
|
||||
live in the parent tests/e2e_tests/conftest.py. PassthroughClient subclasses
|
||||
ProxyClient, so it satisfies lifecycle.ResourceClient and the shared `resources`
|
||||
fixture cleans up keys this suite creates.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from passthrough_client import PassthroughClient, build_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> PassthroughClient:
|
||||
return build_client()
|
||||
|
|
@ -1,136 +0,0 @@
|
|||
"""Client for LLM-translation e2e tests over the proxy's passthrough endpoints.
|
||||
|
||||
Extends the shared ProxyClient with native provider passthrough calls. A
|
||||
passthrough request is sent in the PROVIDER's native format (Gemini
|
||||
generateContent, Anthropic /v1/messages) to the proxy, which forwards it to the
|
||||
provider and still logs a SpendLogs row (call_type="pass_through_endpoint"). The
|
||||
litellm virtual key is passed as the provider key; the proxy swaps in the real
|
||||
env credential. SpendLogs.request_id == the x-litellm-call-id response header.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from proxy_client import ProxyClient, proxy_client_kwargs
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class PassthroughResult:
|
||||
"""Outcome of a native passthrough call. ``call_id`` correlates to the row."""
|
||||
|
||||
status_code: int
|
||||
call_id: Optional[str] # x-litellm-call-id -> SpendLogs.request_id
|
||||
body: str
|
||||
chunks: int = 0 # number of streamed events (0 for non-streaming)
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return 200 <= self.status_code < 300
|
||||
|
||||
|
||||
def _tag_header(tags: Optional[List[str]]) -> dict:
|
||||
return {"tags": ",".join(tags)} if tags else {}
|
||||
|
||||
|
||||
class PassthroughClient(ProxyClient):
|
||||
# ---- Gemini native passthrough (/gemini/v1beta/...) -----------------
|
||||
|
||||
def gemini_generate(
|
||||
self,
|
||||
key: str,
|
||||
model: str,
|
||||
text: str,
|
||||
*,
|
||||
tools: Optional[list] = None,
|
||||
tags: Optional[List[str]] = None,
|
||||
) -> PassthroughResult:
|
||||
body: dict = {"contents": [{"role": "user", "parts": [{"text": text}]}]}
|
||||
if tools is not None:
|
||||
body["tools"] = tools
|
||||
headers = {
|
||||
"x-goog-api-key": key,
|
||||
"Content-Type": "application/json",
|
||||
**_tag_header(tags),
|
||||
}
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/gemini/v1beta/models/{model}:generateContent",
|
||||
headers=headers,
|
||||
json=body,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
return PassthroughResult(
|
||||
resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text
|
||||
)
|
||||
|
||||
def gemini_stream(
|
||||
self, key: str, model: str, text: str, *, tags: Optional[List[str]] = None
|
||||
) -> PassthroughResult:
|
||||
headers = {
|
||||
"x-goog-api-key": key,
|
||||
"Content-Type": "application/json",
|
||||
**_tag_header(tags),
|
||||
}
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/gemini/v1beta/models/{model}:streamGenerateContent",
|
||||
headers=headers,
|
||||
params={"alt": "sse"},
|
||||
json={"contents": [{"role": "user", "parts": [{"text": text}]}]},
|
||||
stream=True,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
call_id = resp.headers.get("x-litellm-call-id")
|
||||
if not (200 <= resp.status_code < 300):
|
||||
return PassthroughResult(resp.status_code, call_id, resp.text)
|
||||
chunks = sum(1 for line in resp.iter_lines() if line)
|
||||
return PassthroughResult(resp.status_code, call_id, "<streamed>", chunks)
|
||||
|
||||
# ---- Anthropic native passthrough (/anthropic/v1/messages) ----------
|
||||
|
||||
def anthropic_message(
|
||||
self,
|
||||
key: str,
|
||||
model: str,
|
||||
text: str,
|
||||
*,
|
||||
max_tokens: int = 64,
|
||||
tools: Optional[list] = None,
|
||||
stream: bool = False,
|
||||
tags: Optional[List[str]] = None,
|
||||
) -> PassthroughResult:
|
||||
body: dict = {
|
||||
"model": model,
|
||||
"max_tokens": max_tokens,
|
||||
"messages": [{"role": "user", "content": text}],
|
||||
}
|
||||
if tools is not None:
|
||||
body["tools"] = tools
|
||||
if stream:
|
||||
body["stream"] = True
|
||||
headers = {
|
||||
"x-api-key": key,
|
||||
"anthropic-version": "2023-06-01",
|
||||
"Content-Type": "application/json",
|
||||
**_tag_header(tags),
|
||||
}
|
||||
url = f"{self._base_url}/anthropic/v1/messages"
|
||||
if not stream:
|
||||
resp = requests.post(
|
||||
url, headers=headers, json=body, timeout=self._request_timeout
|
||||
)
|
||||
return PassthroughResult(
|
||||
resp.status_code, resp.headers.get("x-litellm-call-id"), resp.text
|
||||
)
|
||||
resp = requests.post(
|
||||
url, headers=headers, json=body, stream=True, timeout=self._request_timeout
|
||||
)
|
||||
call_id = resp.headers.get("x-litellm-call-id")
|
||||
if not (200 <= resp.status_code < 300):
|
||||
return PassthroughResult(resp.status_code, call_id, resp.text)
|
||||
chunks = sum(1 for line in resp.iter_lines() if line)
|
||||
return PassthroughResult(resp.status_code, call_id, "<streamed>", chunks)
|
||||
|
||||
|
||||
def build_client() -> PassthroughClient:
|
||||
return PassthroughClient(**proxy_client_kwargs())
|
||||
|
|
@ -1,158 +0,0 @@
|
|||
"""Live e2e for LLM-translation passthrough endpoints.
|
||||
|
||||
Each test sends a NATIVE provider request through the proxy's passthrough route
|
||||
and verifies the proxy still logged a costed SpendLogs row
|
||||
(call_type="pass_through_endpoint"), correlated by the x-litellm-call-id header.
|
||||
|
||||
Covered: gemini ("gemini-2.5-flash") + anthropic ("claude-haiku-4-5"), streaming +
|
||||
non-streaming, plus native tool calls. See LLM_TRANSLATION_COVERAGE_MATRIX.md.
|
||||
|
||||
Skip on environment, fail on behavior: a passthrough call returning non-2xx
|
||||
(provider key missing) skips; once it returns 2xx, a missing/zero-cost row fails.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from passthrough_client import PassthroughClient, PassthroughResult
|
||||
from proxy_client import SpendLogRow, require_successful_call, unique_marker
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
def _f(value: object) -> float:
|
||||
return float(value) if value is not None else 0.0 # type: ignore[arg-type]
|
||||
|
||||
|
||||
def _s(value: object) -> str:
|
||||
return str(value) if value is not None else ""
|
||||
|
||||
|
||||
def _costed_row(client: PassthroughClient, result: PassthroughResult) -> SpendLogRow:
|
||||
"""The passthrough call's logged row, polled until it carries a cost.
|
||||
|
||||
Asserts (not skips) that a 2xx passthrough call produced a costed row - the
|
||||
whole point of passthrough spend tracking.
|
||||
"""
|
||||
assert result.call_id, "passthrough response had no x-litellm-call-id header"
|
||||
rows = client.poll_logs_for_request_id(
|
||||
result.call_id,
|
||||
predicate=lambda rs: _f(rs[0].get("spend")) > 0,
|
||||
)
|
||||
assert rows, f"no SpendLogs row for passthrough call_id {result.call_id}"
|
||||
row = rows[0]
|
||||
assert _s(row.get("call_type")) == "pass_through_endpoint"
|
||||
assert _f(row.get("spend")) > 0, f"passthrough call was not costed: {row}"
|
||||
assert _s(row.get("status")) == "success"
|
||||
return row
|
||||
|
||||
|
||||
# ---- Gemini passthrough ------------------------------------------------
|
||||
|
||||
|
||||
def test_gemini_passthrough_nonstreaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
tag = f"e2e-passthrough-{unique_marker()}"
|
||||
result = client.gemini_generate(
|
||||
scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"]
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "gemini"
|
||||
assert "gemini" in _s(row.get("model"))
|
||||
assert tag in _s(row.get("request_tags")), f"tags not logged: {row.get('request_tags')}"
|
||||
|
||||
|
||||
def test_gemini_passthrough_streaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five")
|
||||
require_successful_call(result)
|
||||
assert result.chunks > 0, "streaming passthrough produced no events"
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "gemini"
|
||||
|
||||
|
||||
def test_gemini_passthrough_tool_call_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.gemini_generate(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
"What is the weather in Paris? Use the get_weather tool.",
|
||||
tools=[
|
||||
{
|
||||
"functionDeclarations": [
|
||||
{
|
||||
"name": "get_weather",
|
||||
"description": "Get the weather for a city",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
"required": ["city"],
|
||||
},
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
)
|
||||
require_successful_call(result)
|
||||
assert "functionCall" in result.body, "gemini did not emit a tool call"
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "gemini"
|
||||
|
||||
|
||||
# ---- Anthropic passthrough ---------------------------------------------
|
||||
|
||||
|
||||
def test_anthropic_passthrough_nonstreaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello")
|
||||
require_successful_call(result)
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "anthropic"
|
||||
assert "claude" in _s(row.get("model"))
|
||||
|
||||
|
||||
def test_anthropic_passthrough_streaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(
|
||||
scoped_key, "claude-haiku-4-5", "Count to five", stream=True
|
||||
)
|
||||
require_successful_call(result)
|
||||
assert result.chunks > 0, "streaming passthrough produced no events"
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "anthropic"
|
||||
|
||||
|
||||
def test_anthropic_passthrough_tool_call_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.anthropic_message(
|
||||
scoped_key,
|
||||
"claude-haiku-4-5",
|
||||
"What is the weather in Paris? Use the get_weather tool.",
|
||||
tools=[
|
||||
{
|
||||
"name": "get_weather",
|
||||
"description": "Get the weather for a city",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
"required": ["city"],
|
||||
},
|
||||
}
|
||||
],
|
||||
)
|
||||
require_successful_call(result)
|
||||
assert "tool_use" in result.body, "anthropic did not emit a tool call"
|
||||
|
||||
row = _costed_row(client, result)
|
||||
assert _s(row.get("custom_llm_provider")) == "anthropic"
|
||||
|
|
@ -1,371 +0,0 @@
|
|||
"""Generic HTTP client for live e2e tests against a running LiteLLM proxy.
|
||||
|
||||
Shared by every e2e suite under tests/e2e_tests/. Covers the proxy operations any
|
||||
suite needs: key/customer management (so the shared ResourceManager can clean up),
|
||||
OpenAI-compatible calls, route probing, and SpendLogs read-back. Suite-specific
|
||||
clients subclass ProxyClient (see spend_tracking/, llm_translation/, budgets/).
|
||||
|
||||
Talks to the proxy over real HTTP so every test sees what a real client sees:
|
||||
the x-litellm-call-id header, the response body, and the rows the proxy writes.
|
||||
Writes are eventually consistent (proxy_batch_write_at ~60s), so read-backs poll
|
||||
to a deadline rather than sleeping once.
|
||||
"""
|
||||
|
||||
import json
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, List, Optional, Protocol, runtime_checkable
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
|
||||
from e2e_config import (
|
||||
MASTER_KEY,
|
||||
POLL_INTERVAL,
|
||||
POLL_TIMEOUT,
|
||||
PROXY_BASE_URL,
|
||||
REQUEST_TIMEOUT,
|
||||
)
|
||||
|
||||
SpendLogRow = Dict[str, object]
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class CallOutcome(Protocol):
|
||||
"""Anything with an HTTP status and body that can pass the skip/fail boundary."""
|
||||
|
||||
status_code: int
|
||||
body: str
|
||||
|
||||
@property
|
||||
def ok(self) -> bool: ...
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ProbeResult:
|
||||
"""Outcome of a single route probe: enough to see *why* it (mis)behaved."""
|
||||
|
||||
url: str
|
||||
status_code: int
|
||||
body: str
|
||||
|
||||
@property
|
||||
def healthy(self) -> bool:
|
||||
# Route exists (not 404), handler did not crash (not 5xx), request
|
||||
# completed (not -1). A 4xx (missing params/auth) still means it ran.
|
||||
return 200 <= self.status_code < 500 and self.status_code != 404
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"GET {self.url} -> {self.status_code}\n{self.body[:600]}"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CallResult:
|
||||
"""Outcome of a single OpenAI-compatible call made through the proxy."""
|
||||
|
||||
status_code: int
|
||||
call_id: Optional[str] # x-litellm-call-id response header
|
||||
response_id: Optional[str] # body "id"; SpendLogs.request_id is derived from this
|
||||
response_cost_header: Optional[str] # x-litellm-response-cost header
|
||||
body: str
|
||||
content: Optional[str]
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return 200 <= self.status_code < 300
|
||||
|
||||
|
||||
def _auth(key: str) -> Dict[str, str]:
|
||||
return {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
|
||||
|
||||
|
||||
class ProxyClient:
|
||||
def __init__(
|
||||
self,
|
||||
base_url: str,
|
||||
master_key: str,
|
||||
*,
|
||||
request_timeout: float,
|
||||
poll_timeout: float,
|
||||
poll_interval: float,
|
||||
) -> None:
|
||||
self._base_url = base_url.rstrip("/")
|
||||
self._master_key = master_key
|
||||
self._request_timeout = request_timeout
|
||||
self._poll_timeout = poll_timeout
|
||||
self._poll_interval = poll_interval
|
||||
|
||||
# ---- key / customer management (satisfies lifecycle.ResourceClient) ----
|
||||
|
||||
def generate_key(
|
||||
self,
|
||||
*,
|
||||
models: Optional[List[str]] = None,
|
||||
max_budget: Optional[float] = None,
|
||||
metadata: Optional[Dict[str, object]] = None,
|
||||
extra_params: Optional[Dict[str, object]] = None,
|
||||
) -> str:
|
||||
payload: Dict[str, object] = {"models": models or [], "duration": None}
|
||||
if max_budget is not None:
|
||||
payload["max_budget"] = max_budget
|
||||
if metadata is not None:
|
||||
payload["metadata"] = metadata
|
||||
if extra_params:
|
||||
payload.update(extra_params)
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/key/generate",
|
||||
headers=_auth(self._master_key),
|
||||
json=payload,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return str(resp.json()["key"])
|
||||
|
||||
def key_info(self, key: str) -> Dict[str, object]:
|
||||
resp = requests.get(
|
||||
f"{self._base_url}/key/info",
|
||||
headers=_auth(self._master_key),
|
||||
params={"key": key},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return dict(resp.json().get("info", {}))
|
||||
|
||||
def delete_key(self, key: str) -> None:
|
||||
"""Best-effort teardown; a failed cleanup must not fail the test."""
|
||||
try:
|
||||
requests.post(
|
||||
f"{self._base_url}/key/delete",
|
||||
headers=_auth(self._master_key),
|
||||
json={"keys": [key]},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
except requests.RequestException:
|
||||
pass
|
||||
|
||||
def delete_customers(self, user_ids: List[str]) -> None:
|
||||
"""Best-effort teardown of end-user/customer rows the `user` param creates."""
|
||||
if not user_ids:
|
||||
return
|
||||
try:
|
||||
requests.post(
|
||||
f"{self._base_url}/customer/delete",
|
||||
headers=_auth(self._master_key),
|
||||
json={"user_ids": user_ids},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
except requests.RequestException:
|
||||
pass
|
||||
|
||||
# ---- OpenAI-compatible calls ----------------------------------------
|
||||
|
||||
def chat(
|
||||
self,
|
||||
key: str,
|
||||
model: str,
|
||||
content: str,
|
||||
*,
|
||||
stream: bool = False,
|
||||
metadata: Optional[Dict[str, object]] = None,
|
||||
extra_body: Optional[Dict[str, object]] = None,
|
||||
) -> CallResult:
|
||||
body: Dict[str, object] = {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": content}],
|
||||
"stream": stream,
|
||||
}
|
||||
if metadata is not None:
|
||||
body["metadata"] = metadata
|
||||
if extra_body is not None:
|
||||
body.update(extra_body)
|
||||
url = f"{self._base_url}/chat/completions"
|
||||
if stream:
|
||||
return self._chat_stream(url, key, body)
|
||||
resp = requests.post(
|
||||
url, headers=_auth(key), json=body, timeout=self._request_timeout
|
||||
)
|
||||
parsed = resp.json() if resp.content else {}
|
||||
choices = parsed.get("choices") or [{}]
|
||||
message_content = (choices[0].get("message") or {}).get("content")
|
||||
return CallResult(
|
||||
status_code=resp.status_code,
|
||||
call_id=resp.headers.get("x-litellm-call-id"),
|
||||
response_id=parsed.get("id"),
|
||||
response_cost_header=resp.headers.get("x-litellm-response-cost"),
|
||||
body=resp.text,
|
||||
content=message_content,
|
||||
)
|
||||
|
||||
def _chat_stream(self, url: str, key: str, body: Dict[str, object]) -> CallResult:
|
||||
resp = requests.post(
|
||||
url,
|
||||
headers=_auth(key),
|
||||
json=body,
|
||||
stream=True,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
if not (200 <= resp.status_code < 300):
|
||||
return CallResult(
|
||||
status_code=resp.status_code,
|
||||
call_id=resp.headers.get("x-litellm-call-id"),
|
||||
response_id=None,
|
||||
response_cost_header=resp.headers.get("x-litellm-response-cost"),
|
||||
body=resp.text,
|
||||
content=None,
|
||||
)
|
||||
response_id: Optional[str] = None
|
||||
parts: List[str] = []
|
||||
for raw in resp.iter_lines():
|
||||
if not raw:
|
||||
continue
|
||||
line = raw.decode("utf-8")
|
||||
if not line.startswith("data:"):
|
||||
continue
|
||||
data = line[len("data:") :].strip()
|
||||
if data == "[DONE]":
|
||||
break
|
||||
chunk = json.loads(data)
|
||||
response_id = chunk.get("id", response_id)
|
||||
for choice in chunk.get("choices", []):
|
||||
piece = (choice.get("delta") or {}).get("content")
|
||||
if piece:
|
||||
parts.append(piece)
|
||||
return CallResult(
|
||||
status_code=resp.status_code,
|
||||
call_id=resp.headers.get("x-litellm-call-id"),
|
||||
response_id=response_id,
|
||||
response_cost_header=resp.headers.get("x-litellm-response-cost"),
|
||||
body="<streamed>",
|
||||
content="".join(parts) or None,
|
||||
)
|
||||
|
||||
def embed(self, key: str, model: str, text: str) -> CallResult:
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/embeddings",
|
||||
headers=_auth(key),
|
||||
json={"model": model, "input": text},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
parsed = resp.json() if resp.content else {}
|
||||
return CallResult(
|
||||
status_code=resp.status_code,
|
||||
call_id=resp.headers.get("x-litellm-call-id"),
|
||||
response_id=parsed.get("id"),
|
||||
response_cost_header=resp.headers.get("x-litellm-response-cost"),
|
||||
body=resp.text,
|
||||
content=None,
|
||||
)
|
||||
|
||||
# ---- route discovery -------------------------------------------------
|
||||
|
||||
def get_openapi(self) -> Dict[str, object]:
|
||||
"""The proxy's live route schema from /openapi.json."""
|
||||
resp = requests.get(
|
||||
f"{self._base_url}/openapi.json", timeout=self._request_timeout
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return dict(resp.json())
|
||||
|
||||
def probe(self, path: str, params: Optional[Dict[str, str]] = None) -> ProbeResult:
|
||||
"""GET a route with master-key auth; capture status + body to show why."""
|
||||
url = f"{self._base_url}{path}"
|
||||
try:
|
||||
resp = requests.get(
|
||||
url,
|
||||
headers=_auth(self._master_key),
|
||||
params=params or {},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
except requests.RequestException as exc:
|
||||
return ProbeResult(url=url, status_code=-1, body=f"request error: {exc}")
|
||||
return ProbeResult(url=url, status_code=resp.status_code, body=resp.text)
|
||||
|
||||
# ---- SpendLogs read-back --------------------------------------------
|
||||
|
||||
def _get_logs(
|
||||
self, *, request_id: Optional[str] = None, api_key: Optional[str] = None
|
||||
) -> List[SpendLogRow]:
|
||||
params: Dict[str, str] = {}
|
||||
if request_id is not None:
|
||||
params["request_id"] = request_id
|
||||
if api_key is not None:
|
||||
params["api_key"] = api_key
|
||||
resp = requests.get(
|
||||
f"{self._base_url}/spend/logs",
|
||||
headers=_auth(self._master_key),
|
||||
params=params,
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
return []
|
||||
data = resp.json()
|
||||
return [dict(row) for row in data] if isinstance(data, list) else []
|
||||
|
||||
def poll_logs_for_key(
|
||||
self,
|
||||
key: str,
|
||||
*,
|
||||
min_rows: int = 1,
|
||||
predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None,
|
||||
) -> List[SpendLogRow]:
|
||||
return self._poll(lambda: self._get_logs(api_key=key), min_rows, predicate)
|
||||
|
||||
def poll_logs_for_request_id(
|
||||
self,
|
||||
request_id: str,
|
||||
*,
|
||||
min_rows: int = 1,
|
||||
predicate: Optional[Callable[[List[SpendLogRow]], bool]] = None,
|
||||
) -> List[SpendLogRow]:
|
||||
return self._poll(
|
||||
lambda: self._get_logs(request_id=request_id), min_rows, predicate
|
||||
)
|
||||
|
||||
def _poll(
|
||||
self,
|
||||
fetch: Callable[[], List[SpendLogRow]],
|
||||
min_rows: int,
|
||||
predicate: Optional[Callable[[List[SpendLogRow]], bool]],
|
||||
) -> List[SpendLogRow]:
|
||||
deadline = time.monotonic() + self._poll_timeout
|
||||
rows: List[SpendLogRow] = []
|
||||
while time.monotonic() < deadline:
|
||||
rows = fetch()
|
||||
satisfied = len(rows) >= min_rows and (
|
||||
predicate is None or predicate(rows)
|
||||
)
|
||||
if satisfied:
|
||||
return rows
|
||||
time.sleep(self._poll_interval)
|
||||
return rows
|
||||
|
||||
|
||||
def proxy_client_kwargs() -> Dict[str, object]:
|
||||
"""Constructor kwargs shared by every ProxyClient subclass."""
|
||||
return {
|
||||
"base_url": PROXY_BASE_URL,
|
||||
"master_key": MASTER_KEY,
|
||||
"request_timeout": REQUEST_TIMEOUT,
|
||||
"poll_timeout": POLL_TIMEOUT,
|
||||
"poll_interval": POLL_INTERVAL,
|
||||
}
|
||||
|
||||
|
||||
def require_successful_call(result: CallOutcome) -> None:
|
||||
"""The skip/fail boundary.
|
||||
|
||||
A non-2xx here means the environment cannot make the call (missing provider
|
||||
key, upstream down) -> skip. Everything downstream of a 2xx is behavior that
|
||||
is allowed to fail.
|
||||
"""
|
||||
if result.ok:
|
||||
return
|
||||
pytest.skip(
|
||||
f"upstream call unavailable (status {result.status_code}); "
|
||||
f"body={result.body[:300]}"
|
||||
)
|
||||
|
||||
|
||||
def unique_marker() -> str:
|
||||
return uuid.uuid4().hex[:12]
|
||||
|
|
@ -1,7 +0,0 @@
|
|||
[pytest]
|
||||
# Config when any e2e suite under tests/e2e_tests/ is run directly, e.g.
|
||||
# uv run pytest tests/e2e_tests/spend_tracking/ -v
|
||||
# The e2e marker is also registered in conftest.py for runs rooted elsewhere.
|
||||
addopts = --strict-markers --strict-config
|
||||
markers =
|
||||
e2e: live test that requires a running proxy and real provider keys
|
||||
|
|
@ -1,77 +0,0 @@
|
|||
# Spend Tracking Test Coverage Matrix
|
||||
|
||||
Scope: every distinct spend-tracking code path, mapped to the test that exercises
|
||||
it and the level it runs at. Highlights where a live e2e check is the only thing
|
||||
that would catch a regression.
|
||||
|
||||
Companion: live suite `test_spend_tracking_e2e.py` + route breadth
|
||||
`test_spend_routes.py` (this directory). Offline regression suite:
|
||||
`tests/test_litellm/proxy/spend_tracking/`. Reference PR: BerriAI/litellm#29956.
|
||||
|
||||
Levels: `unit` mocked; `integration` real DB/cost-map; `live` real provider +
|
||||
proxy + SpendLogs rows. Status: `covered` / `partial` / `gap`.
|
||||
|
||||
---
|
||||
|
||||
## SpendLogs row construction (`spend_tracking_utils.get_logging_payload`)
|
||||
|
||||
| Path | Existing | Level | Status | Live e2e |
|
||||
|------|----------|-------|--------|----------|
|
||||
| `_get_status_for_spend_log` | `test_spend_tracking_utils.py` | unit | covered | yes (status read off the row) |
|
||||
| cache-hit `request_id` suffix | `test_spend_tracking_utils.py` | unit | covered | yes (`test_cache_hit_is_zero_cost_and_suffixed`) |
|
||||
| failure status + zero spend | `test_spend_tracking_utils.py` | unit | partial | yes (`test_failure_call_writes_failure_status_row`) |
|
||||
| field population (model/tokens/api_key/team/org) | `test_spend_tracking_utils.py` | unit | partial | yes (asserts real values) |
|
||||
| `request_tags` propagation | `test_db_spend_update_writer.py` | unit | partial | yes (`test_request_tags_round_trip`) |
|
||||
| `end_user` attribution | unit | unit | partial | yes (`test_end_user_spend_attributed_on_row`) |
|
||||
|
||||
## Cost calculation by modality
|
||||
|
||||
| Modality | Existing | Status | Live e2e |
|
||||
|----------|----------|--------|----------|
|
||||
| Chat (non-stream) | `test_cost_calculator.py`, `local_testing/test_completion_cost.py` | covered | yes (`test_chat_completion_writes_nonzero_spend_row`) |
|
||||
| Chat (streaming) | `test_streaming_interrupt_spend_tracking.py` | partial | yes (`test_streaming_chat_completion_tracks_spend`) |
|
||||
| Embedding | `test_cost_calculator.py` (#29956) | partial | yes (`test_embedding_writes_nonzero_spend_row`) |
|
||||
| Pass-through (gemini/anthropic) | `pass_through_tests/*.test.js` + `llm_translation/` suite | covered | yes (llm_translation suite) |
|
||||
| Image / audio / rerank / responses / realtime | per-provider unit cost tests | partial/gap | gap |
|
||||
|
||||
## Entity spend aggregation
|
||||
|
||||
| Entity | Existing | Status | Live e2e |
|
||||
|--------|----------|--------|----------|
|
||||
| API key | `test_db_spend_update_writer.py`, `test_spend_counters.py` | covered | yes (`test_key_spend_equals_sum_of_logs`) |
|
||||
| Tag | `test_update_daily_tag_spend.py` | partial | yes (`test_tag_spend_matches_sum_of_tagged_logs`) |
|
||||
| End-user | `test_proxy_update_spend.py` | covered | yes |
|
||||
| Spend == sum(logs) consistency | none | gap | yes (key + tag aggregate == sum of rows) |
|
||||
|
||||
## Spend read endpoints (verification surface)
|
||||
|
||||
| Endpoint | Existing | Status | Live e2e |
|
||||
|----------|----------|--------|----------|
|
||||
| `/spend/logs` (request_id / api_key) | `test_spend_management_endpoints.py` | covered | yes (primary read path) |
|
||||
| `/spend/calculate` | `local_testing/test_spend_calculate_endpoint.py` | covered | yes (`test_spend_calculate_returns_nonzero_cost`) |
|
||||
| `/spend/tags` | `test_spend_management_endpoints.py` | partial | yes (tag accuracy test) |
|
||||
| whole spend GET surface (22 routes) | unit per-handler | partial | yes (`test_spend_routes.py` probes each for 404/5xx) |
|
||||
|
||||
## What this suite pins
|
||||
|
||||
| Test | Invariant |
|
||||
|------|-----------|
|
||||
| `test_chat_completion_writes_nonzero_spend_row` | nonzero cost, token arithmetic, status, row findable by `response.id` |
|
||||
| `test_streaming_chat_completion_tracks_spend` | streamed responses still costed |
|
||||
| `test_embedding_writes_nonzero_spend_row` | embedding cost, `completion_tokens == 0` |
|
||||
| `test_cache_hit_is_zero_cost_and_suffixed` | cache hits not double-charged; `_cache_hit` suffix |
|
||||
| `test_key_spend_equals_sum_of_logs` | key aggregate == sum of rows |
|
||||
| `test_request_tags_round_trip` | tags persist onto the row |
|
||||
| `test_tag_spend_matches_sum_of_tagged_logs` | `/spend/tags` SUM/COUNT == tagged rows |
|
||||
| `test_end_user_spend_attributed_on_row` | `end_user` attributed + costed |
|
||||
| `test_failure_call_writes_failure_status_row` | failed call -> `status=failure`, `spend=0` |
|
||||
| `test_spend_calculate_returns_nonzero_cost` | cost-map smoke (no batch wait) |
|
||||
| `test_spend_routes.py` (23) | no spend route 404s or 5xxs |
|
||||
|
||||
## Design + timing
|
||||
|
||||
`proxy_batch_write_at` (~60s) means rows land late; every read polls to a deadline.
|
||||
Fresh scoped key per test (isolation, xdist-safe, cleaned up). Assert invariants
|
||||
(`spend > 0`, `total == prompt + completion`, aggregate == sum), not literal
|
||||
$/token values, so pricing drift is not a failure. Skip on environment (no proxy /
|
||||
no provider key), fail on behavior (a real 2xx call with a wrong/missing row).
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
"""Spend-tracking suite's `client` fixture.
|
||||
|
||||
The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
|
||||
live in the parent tests/e2e_tests/conftest.py. SpendE2EClient satisfies the
|
||||
lifecycle.ResourceClient protocol, so the shared `resources` fixture cleans up
|
||||
keys and customers this suite creates.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from spend_e2e_client import SpendE2EClient, build_client
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> SpendE2EClient:
|
||||
return build_client()
|
||||
|
|
@ -1,94 +0,0 @@
|
|||
"""Spend-tracking e2e client: the generic ProxyClient plus spend-specific reads.
|
||||
|
||||
Generic proxy operations (keys, customers, chat/embed, route probing, SpendLogs
|
||||
polling) live in the shared tests/e2e_tests/proxy_client.py. This module adds only
|
||||
the spend-specific endpoints: /spend/calculate, /spend/tags, and key-spend polling.
|
||||
|
||||
Re-exports CallResult / SpendLogRow / require_successful_call / unique_marker so
|
||||
existing tests keep importing them from here.
|
||||
"""
|
||||
|
||||
import time
|
||||
from typing import List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from proxy_client import (
|
||||
CallResult,
|
||||
ProbeResult,
|
||||
ProxyClient,
|
||||
SpendLogRow,
|
||||
_auth,
|
||||
proxy_client_kwargs,
|
||||
require_successful_call,
|
||||
unique_marker,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"SpendE2EClient",
|
||||
"build_client",
|
||||
"CallResult",
|
||||
"ProbeResult",
|
||||
"SpendLogRow",
|
||||
"require_successful_call",
|
||||
"unique_marker",
|
||||
]
|
||||
|
||||
|
||||
class SpendE2EClient(ProxyClient):
|
||||
def calculate_spend(self, model: str, content: str) -> float:
|
||||
resp = requests.post(
|
||||
f"{self._base_url}/spend/calculate",
|
||||
headers=_auth(self._master_key),
|
||||
json={"model": model, "messages": [{"role": "user", "content": content}]},
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return float(resp.json()["cost"])
|
||||
|
||||
def spend_by_tags(self) -> List[SpendLogRow]:
|
||||
"""/spend/tags: SUM(spend) GROUP BY tag, straight from SpendLogs."""
|
||||
resp = requests.get(
|
||||
f"{self._base_url}/spend/tags",
|
||||
headers=_auth(self._master_key),
|
||||
timeout=self._request_timeout,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
return []
|
||||
data = resp.json()
|
||||
if isinstance(data, dict) and "spend_per_tag" in data:
|
||||
data = data["spend_per_tag"]
|
||||
return [dict(row) for row in data] if isinstance(data, list) else []
|
||||
|
||||
def poll_tag_spend(
|
||||
self, tag: str, *, minimum: float = 0.0
|
||||
) -> Optional[SpendLogRow]:
|
||||
"""Poll until the tag's aggregate spend reaches `minimum`; last seen entry."""
|
||||
deadline = time.monotonic() + self._poll_timeout
|
||||
entry: Optional[SpendLogRow] = None
|
||||
while time.monotonic() < deadline:
|
||||
matches = [
|
||||
row
|
||||
for row in self.spend_by_tags()
|
||||
if str(row.get("individual_request_tag")) == tag
|
||||
]
|
||||
if matches:
|
||||
entry = matches[0]
|
||||
if float(entry.get("total_spend") or 0.0) >= minimum:
|
||||
return entry
|
||||
time.sleep(self._poll_interval)
|
||||
return entry
|
||||
|
||||
def poll_key_spend(self, key: str, *, minimum: float = 0.0) -> float:
|
||||
deadline = time.monotonic() + self._poll_timeout
|
||||
spend = 0.0
|
||||
while time.monotonic() < deadline:
|
||||
spend = float(self.key_info(key).get("spend") or 0.0)
|
||||
if spend > minimum:
|
||||
return spend
|
||||
time.sleep(self._poll_interval)
|
||||
return spend
|
||||
|
||||
|
||||
def build_client() -> SpendE2EClient:
|
||||
return SpendE2EClient(**proxy_client_kwargs())
|
||||
|
|
@ -1,93 +0,0 @@
|
|||
"""Breadth check: query every route on the spend read surface and show what it
|
||||
returns.
|
||||
|
||||
Spend tracking sprawls across many routes (model-cost / key / user / team / org /
|
||||
customer aggregation, tags, and activity reports). Most are served with
|
||||
`include_in_schema=False`, so they do NOT appear in `/openapi.json` - discovery
|
||||
from the schema alone misses ~70% of the surface. So we probe a curated, verified
|
||||
list directly, plus any spend route the schema does list (to auto-catch new ones).
|
||||
|
||||
Each probe captures status AND body, so a failure shows the proxy's actual error
|
||||
(a 500 traceback, a 404 meaning the route was removed) rather than a bare code.
|
||||
Run with `-rA` (or `-s`) to print every route's response, not just failures.
|
||||
|
||||
Healthy == route exists (not 404) and handler did not crash (not 5xx). A 4xx
|
||||
(missing params / auth nuance) still means the route is wired and ran. Cheap and
|
||||
fast: no batch-write wait, no provider calls.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
from spend_e2e_client import SpendE2EClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
# Verified present and responsive on a live proxy. One per row of the spend
|
||||
# surface: key / user / team / org / customer aggregation, model-cost, tags,
|
||||
# activity.
|
||||
SPEND_ROUTES = (
|
||||
"/spend/keys",
|
||||
"/spend/users",
|
||||
"/spend/tags",
|
||||
"/spend/logs",
|
||||
"/spend/logs/ui",
|
||||
"/global/spend",
|
||||
"/global/spend/keys",
|
||||
"/global/spend/teams",
|
||||
"/global/spend/models",
|
||||
"/global/spend/provider",
|
||||
"/global/spend/report",
|
||||
"/global/spend/tags",
|
||||
"/global/spend/logs",
|
||||
"/global/spend/all_tag_names",
|
||||
"/global/activity",
|
||||
"/global/activity/model",
|
||||
"/global/activity/exceptions",
|
||||
"/key/list",
|
||||
"/user/list",
|
||||
"/team/list",
|
||||
"/organization/list",
|
||||
"/customer/list",
|
||||
)
|
||||
|
||||
_SPEND_PREFIXES = ("/spend", "/global/spend", "/global/activity")
|
||||
|
||||
|
||||
def _default_params() -> Dict[str, str]:
|
||||
# Satisfies date-required endpoints (report/activity/provider); ignored elsewhere.
|
||||
end = datetime.now(timezone.utc).date()
|
||||
start = end - timedelta(days=1)
|
||||
return {"start_date": start.isoformat(), "end_date": end.isoformat()}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("route", SPEND_ROUTES)
|
||||
def test_spend_route_responsive(client: SpendE2EClient, route: str) -> None:
|
||||
result = client.probe(route, params=_default_params())
|
||||
print(result) # shown on failure, and for all routes under `-rA` / `-s`
|
||||
assert result.healthy, str(result)
|
||||
|
||||
|
||||
def test_schema_listed_spend_routes_are_responsive(client: SpendE2EClient) -> None:
|
||||
"""Probe any spend GET route the schema lists that isn't in SPEND_ROUTES."""
|
||||
paths = client.get_openapi().get("paths", {})
|
||||
assert isinstance(paths, dict) and paths, "/openapi.json had no paths"
|
||||
|
||||
discovered: List[str] = [
|
||||
path
|
||||
for path, operations in paths.items()
|
||||
if isinstance(operations, dict)
|
||||
and "get" in {m.lower() for m in operations}
|
||||
and "{" not in path
|
||||
and any(path.startswith(prefix) for prefix in _SPEND_PREFIXES)
|
||||
]
|
||||
extras = [path for path in discovered if path not in SPEND_ROUTES]
|
||||
|
||||
params = _default_params()
|
||||
results = [client.probe(path, params=params) for path in extras]
|
||||
for result in results:
|
||||
print(result)
|
||||
offenders = [str(result) for result in results if not result.healthy]
|
||||
assert not offenders, "non-responsive schema spend routes:\n" + "\n".join(offenders)
|
||||
|
|
@ -1,335 +0,0 @@
|
|||
"""Live end-to-end spend-tracking tests against a running proxy.
|
||||
|
||||
Run against a proxy started with tests/e2e_tests/docker-config.yaml (or the
|
||||
gateway config). Coverage rationale: SPEND_TRACKING_COVERAGE_MATRIX.md.
|
||||
|
||||
Model names are literals from that config: chat tests hit "gemini-2.5-flash",
|
||||
embedding tests hit "openai-text-embedding-3-small".
|
||||
|
||||
Every test: fresh scoped key (isolation) -> real provider call ->
|
||||
require_successful_call (skip iff env can't make it) -> poll /spend/logs to a
|
||||
deadline (rows land ~60s later via proxy_batch_write_at) -> assert invariants on
|
||||
the real row (spend, token arithmetic, status, cache).
|
||||
|
||||
Assertions target invariants, not literals: a regression in the spend pipeline
|
||||
fails the test; a pricing or token-count drift does not.
|
||||
"""
|
||||
|
||||
from typing import Callable, Dict, List
|
||||
|
||||
import pytest
|
||||
|
||||
from lifecycle import ResourceManager
|
||||
from spend_e2e_client import (
|
||||
SpendE2EClient,
|
||||
SpendLogRow,
|
||||
require_successful_call,
|
||||
unique_marker,
|
||||
)
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
def _f(value: object) -> float:
|
||||
return float(value) if value is not None else 0.0 # type: ignore[arg-type]
|
||||
|
||||
|
||||
def _i(value: object) -> int:
|
||||
return int(value) if value is not None else 0 # type: ignore[arg-type]
|
||||
|
||||
|
||||
def _s(value: object) -> str:
|
||||
return str(value) if value is not None else ""
|
||||
|
||||
|
||||
def _summarize(rows: List[SpendLogRow]) -> List[Dict[str, object]]:
|
||||
keys = (
|
||||
"request_id",
|
||||
"model",
|
||||
"spend",
|
||||
"status",
|
||||
"cache_hit",
|
||||
"prompt_tokens",
|
||||
"completion_tokens",
|
||||
"total_tokens",
|
||||
)
|
||||
return [{k: r.get(k) for k in keys} for r in rows]
|
||||
|
||||
|
||||
def _require_row(
|
||||
rows: List[SpendLogRow], predicate: Callable[[SpendLogRow], bool], what: str
|
||||
) -> SpendLogRow:
|
||||
matches = [r for r in rows if predicate(r)]
|
||||
assert matches, (
|
||||
f"no SpendLogs row {what} after polling; saw {len(rows)} row(s): "
|
||||
f"{_summarize(rows)}"
|
||||
)
|
||||
return matches[0]
|
||||
|
||||
|
||||
def test_chat_completion_writes_nonzero_spend_row(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
f"reply with one word {unique_marker()}",
|
||||
extra_body={"max_tokens": 16},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(_s(r.get("status")) == "success" for r in rs),
|
||||
)
|
||||
row = _require_row(
|
||||
rows, lambda r: _s(r.get("status")) == "success", "for the chat call"
|
||||
)
|
||||
|
||||
assert _f(row.get("spend")) > 0, f"chat row should cost > 0: {_summarize(rows)}"
|
||||
assert _s(row.get("status")) == "success"
|
||||
assert _s(row.get("cache_hit")) != "True", "fresh call must not be a cache hit"
|
||||
assert "gemini-2.5-flash" in _s(row.get("model"))
|
||||
|
||||
prompt = _i(row.get("prompt_tokens"))
|
||||
completion = _i(row.get("completion_tokens"))
|
||||
total = _i(row.get("total_tokens"))
|
||||
assert prompt > 0 and completion > 0
|
||||
assert total == prompt + completion, f"token arithmetic broken: {_summarize(rows)}"
|
||||
|
||||
if result.response_id:
|
||||
assert any(
|
||||
_s(r.get("request_id")) == result.response_id for r in rows
|
||||
), f"row request_id != client response.id ({result.response_id})"
|
||||
|
||||
|
||||
def test_streaming_chat_completion_tracks_spend(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
f"count to three {unique_marker()}",
|
||||
stream=True,
|
||||
extra_body={"max_tokens": 64},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs),
|
||||
)
|
||||
row = _require_row(
|
||||
rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the stream"
|
||||
)
|
||||
prompt = _i(row.get("prompt_tokens"))
|
||||
completion = _i(row.get("completion_tokens"))
|
||||
assert prompt > 0 and completion > 0, f"streaming tokens not tracked: {_summarize(rows)}"
|
||||
assert _i(row.get("total_tokens")) == prompt + completion
|
||||
|
||||
|
||||
def test_embedding_writes_nonzero_spend_row(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.embed(
|
||||
scoped_key,
|
||||
"openai-text-embedding-3-small",
|
||||
f"vectorize this sentence {unique_marker()}",
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key, predicate=lambda rs: any(_f(r.get("spend")) > 0 for r in rs)
|
||||
)
|
||||
row = _require_row(
|
||||
rows, lambda r: _f(r.get("spend")) > 0, "with nonzero spend for the embedding"
|
||||
)
|
||||
assert _i(row.get("prompt_tokens")) > 0
|
||||
assert _i(row.get("completion_tokens")) == 0, "embeddings have no completion tokens"
|
||||
assert "text-embedding-3-small" in _s(row.get("model"))
|
||||
|
||||
|
||||
def test_cache_hit_is_zero_cost_and_suffixed(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
# Unique marker shared by both calls: call 1 is a guaranteed cache MISS (fresh
|
||||
# content, paid), call 2 repeats the identical request and HITS the cache just
|
||||
# populated. The marker keeps each run isolated - a fixed prompt would persist
|
||||
# in the shared response cache across runs and make both calls hit (flaky).
|
||||
prompt = f"What is the capital of France? Answer in one word. {unique_marker()}"
|
||||
first = client.chat(
|
||||
scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16}
|
||||
)
|
||||
require_successful_call(first)
|
||||
second = client.chat(
|
||||
scoped_key, "gemini-2.5-flash", prompt, extra_body={"max_tokens": 16}
|
||||
)
|
||||
require_successful_call(second)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(_s(r.get("cache_hit")) == "True" for r in rs),
|
||||
)
|
||||
cache_rows = [r for r in rows if _s(r.get("cache_hit")) == "True"]
|
||||
if not cache_rows:
|
||||
pytest.skip(
|
||||
"no cache-hit row observed; caching may be disabled on this proxy. "
|
||||
f"rows seen: {_summarize(rows)}"
|
||||
)
|
||||
|
||||
cache_row = cache_rows[0]
|
||||
assert _f(cache_row.get("spend")) == 0.0, (
|
||||
f"cache hit was charged (double-charge regression): {_summarize(rows)}"
|
||||
)
|
||||
assert "_cache_hit" in _s(cache_row.get("request_id")), (
|
||||
"cache-hit row missing the _cache_hit request_id suffix; "
|
||||
"duplicate-key collisions will silently drop rows"
|
||||
)
|
||||
paid_rows = [r for r in rows if _s(r.get("cache_hit")) != "True"]
|
||||
assert any(_f(r.get("spend")) > 0 for r in paid_rows), (
|
||||
f"the non-cached call should still be charged: {_summarize(rows)}"
|
||||
)
|
||||
|
||||
|
||||
def test_key_spend_equals_sum_of_logs(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
for _ in range(2):
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
f"say hi {unique_marker()}",
|
||||
extra_body={"max_tokens": 16},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
min_rows=2,
|
||||
predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0,
|
||||
)
|
||||
assert len(rows) >= 2, f"expected >=2 rows for the key, saw {_summarize(rows)}"
|
||||
logs_total = sum(_f(r.get("spend")) for r in rows)
|
||||
assert logs_total > 0
|
||||
|
||||
key_spend = client.poll_key_spend(scoped_key, minimum=logs_total * 0.999)
|
||||
assert key_spend == pytest.approx(logs_total, rel=1e-2, abs=1e-9), (
|
||||
f"key aggregate {key_spend} != sum of logs {logs_total}; "
|
||||
f"rows: {_summarize(rows)}"
|
||||
)
|
||||
|
||||
|
||||
def test_request_tags_round_trip(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
tag = f"e2e-spend-{unique_marker()}"
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
"tagged request",
|
||||
metadata={"tags": [tag]},
|
||||
extra_body={"max_tokens": 16},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(tag in _s(r.get("request_tags")) for r in rs),
|
||||
)
|
||||
_require_row(
|
||||
rows,
|
||||
lambda r: tag in _s(r.get("request_tags")),
|
||||
f"carrying request tag {tag!r}",
|
||||
)
|
||||
|
||||
|
||||
def test_tag_spend_matches_sum_of_tagged_logs(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
# Unique tag so /spend/tags can't be polluted by other rows; unique content
|
||||
# per call so both are fresh misses (paid), not cache hits.
|
||||
tag = f"e2e-tagspend-{unique_marker()}"
|
||||
for _ in range(2):
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
f"hi {unique_marker()}",
|
||||
metadata={"tags": [tag]},
|
||||
extra_body={"max_tokens": 16},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
min_rows=2,
|
||||
predicate=lambda rs: sum(_f(r.get("spend")) for r in rs) > 0,
|
||||
)
|
||||
tagged = [r for r in rows if tag in _s(r.get("request_tags"))]
|
||||
assert len(tagged) >= 2, f"expected 2 tagged rows, saw {_summarize(rows)}"
|
||||
logs_total = sum(_f(r.get("spend")) for r in tagged)
|
||||
assert logs_total > 0
|
||||
|
||||
entry = client.poll_tag_spend(tag, minimum=logs_total * 0.999)
|
||||
assert entry is not None, f"tag {tag!r} never appeared in /spend/tags"
|
||||
assert _f(entry.get("total_spend")) == pytest.approx(
|
||||
logs_total, rel=1e-2, abs=1e-9
|
||||
), f"/spend/tags total_spend {entry} != sum of tagged rows {logs_total}"
|
||||
assert _i(entry.get("log_count")) == len(tagged), (
|
||||
f"/spend/tags log_count {entry.get('log_count')} != tagged rows {len(tagged)}"
|
||||
)
|
||||
|
||||
|
||||
def test_end_user_spend_attributed_on_row(
|
||||
client: SpendE2EClient, scoped_key: str, resources: ResourceManager
|
||||
) -> None:
|
||||
customer = resources.customer(f"e2e-cust-{unique_marker()}")
|
||||
result = client.chat(
|
||||
scoped_key,
|
||||
"gemini-2.5-flash",
|
||||
"hi",
|
||||
extra_body={"user": customer, "max_tokens": 16},
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(_s(r.get("end_user")) == customer for r in rs),
|
||||
)
|
||||
row = _require_row(
|
||||
rows,
|
||||
lambda r: _s(r.get("end_user")) == customer,
|
||||
f"attributed to end_user {customer!r}",
|
||||
)
|
||||
assert _f(row.get("spend")) > 0, f"end-user row should cost > 0: {_summarize(rows)}"
|
||||
|
||||
|
||||
def test_failure_call_writes_failure_status_row(
|
||||
client: SpendE2EClient, scoped_key: str
|
||||
) -> None:
|
||||
result = client.chat(
|
||||
scoped_key, "gemini-2.5-flash", "", extra_body={"max_tokens": 1}
|
||||
)
|
||||
if result.ok:
|
||||
pytest.skip("call unexpectedly succeeded; could not induce a failure row")
|
||||
|
||||
rows = client.poll_logs_for_key(
|
||||
scoped_key,
|
||||
predicate=lambda rs: any(_s(r.get("status")) == "failure" for r in rs),
|
||||
)
|
||||
failure_rows = [r for r in rows if _s(r.get("status")) == "failure"]
|
||||
if not failure_rows:
|
||||
pytest.skip(
|
||||
"no failure-status row was logged for the rejected call; "
|
||||
"failure logging is environment-specific"
|
||||
)
|
||||
assert _f(failure_rows[0].get("spend")) == 0.0, "failed call must not be charged"
|
||||
|
||||
|
||||
def test_spend_calculate_returns_nonzero_cost(client: SpendE2EClient) -> None:
|
||||
cost = client.calculate_spend(
|
||||
"gemini-2.5-flash", "estimate the cost of this request"
|
||||
)
|
||||
assert cost > 0, (
|
||||
"/spend/calculate returned 0 for gemini-2.5-flash; "
|
||||
"cost map may be missing this model"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue