mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-13 23:11:40 +00:00
Introduce the e2e coverage denominator: 282 behavior cells across the six tracking modules (LLMs, MCPs, Management/UI, Reliability & Performance, Logging & Guardrails, Other), one validated YAML row each, plus a collector that diffs the registry against @pytest.mark.covers markers and reports coverage per module. The registry rows validate against a pydantic discriminated union so a row cannot carry a field from another module. The collector is static: a collect-only pass reads the markers, so it runs no test and needs no live proxy. Register the covers marker suite-wide so that pass works under --strict-markers. This is a draft for review. Tiers are proposed rather than signed off, and a few cells still need a support check or a prune.
124 lines
4.8 KiB
Python
124 lines
4.8 KiB
Python
"""Shared fixtures for all live e2e suites under tests/e2e/.
|
|
|
|
Design rule: skip on environment, fail on behavior. Live tests (marked `e2e`)
|
|
skip when no proxy answers; once a request reaches the proxy, behavior is
|
|
asserted. Pure unit coverage of the harness itself carries no `e2e` marker and
|
|
runs regardless of whether a proxy is up.
|
|
|
|
Lifecycle: the `resources` fixture maps the init -> run -> teardown contract
|
|
(lifecycle.E2ECase) onto pytest - setup is init(), the test body is run(), and
|
|
teardown deletes every resource the test created on the long-lived proxy.
|
|
|
|
Each suite provides its own `client` fixture (a lifecycle.ResourceClient); these
|
|
shared fixtures build on it.
|
|
"""
|
|
|
|
import functools
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Iterator
|
|
|
|
import pytest
|
|
import requests
|
|
|
|
from e2e_config import CONTROL_PLANE_BASE_URL, PROXY_BASE_URL
|
|
from lifecycle import GatewayProvider, ResourceManager
|
|
|
|
|
|
_E2E_TEST_RAN = pytest.StashKey[bool]()
|
|
|
|
|
|
def pytest_configure(config: pytest.Config) -> None:
|
|
config.addinivalue_line(
|
|
"markers",
|
|
"e2e: live test that requires a running proxy and real provider keys",
|
|
)
|
|
config.addinivalue_line(
|
|
"markers",
|
|
"covers(cell_id, *, exercised_on=()): coverage-registry cell(s) this test covers",
|
|
)
|
|
|
|
|
|
def _liveness_reason(label: str, base_url: str) -> str | None:
|
|
"""None if `base_url` answers its liveness probe, else a skip reason."""
|
|
try:
|
|
resp = requests.get(f"{base_url}/health/liveliness", timeout=5)
|
|
except requests.RequestException as exc:
|
|
return f"No live {label} at {base_url}: {exc}"
|
|
if resp.status_code >= 500:
|
|
return f"{label} at {base_url} returned {resp.status_code}"
|
|
return None
|
|
|
|
|
|
@functools.lru_cache(maxsize=1)
|
|
def _proxy_skip_reason() -> str | None:
|
|
"""Probe the proxy once per session. None if it answers, else a skip reason. In
|
|
a split deployment the management/admin control plane is a separate service, so
|
|
require it too (when it differs) - else its tests would fail rather than skip."""
|
|
reason = _liveness_reason("proxy", PROXY_BASE_URL)
|
|
if reason is not None:
|
|
return reason
|
|
if CONTROL_PLANE_BASE_URL != PROXY_BASE_URL:
|
|
return _liveness_reason("control plane", CONTROL_PLANE_BASE_URL)
|
|
return None
|
|
|
|
|
|
def pytest_runtest_setup(item: pytest.Item) -> None:
|
|
"""Skip `e2e`-marked tests unless a proxy answers its liveness probe. Unmarked
|
|
tests (unit coverage of the harness) don't touch the proxy, so they run even
|
|
when none is up."""
|
|
if item.get_closest_marker("e2e") is None:
|
|
return
|
|
reason = _proxy_skip_reason()
|
|
if reason is not None:
|
|
pytest.skip(reason)
|
|
|
|
|
|
def pytest_runtest_call(item: pytest.Item) -> None:
|
|
"""Mark that an e2e test body actually ran (not skipped at setup). Skipped
|
|
sessions never reach this hook, so the session-finish cleanup can use it as a
|
|
guard before truncating the spend-log DB. Tests under `tests/e2e/` without the
|
|
`e2e` marker (pure unit coverage for the harness itself) never hit the proxy,
|
|
so they must not arm the destructive DB truncate."""
|
|
if item.get_closest_marker("e2e") is None:
|
|
return
|
|
item.session.stash[_E2E_TEST_RAN] = True
|
|
|
|
|
|
def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None:
|
|
"""Once the whole e2e session is done (all suites), truncate the spend logs so
|
|
the DB doesn't accumulate test rows. Skipped sessions (no live proxy, no test
|
|
actually executed) leave the DB alone so a `DATABASE_URL` pointing at a shared
|
|
instance is never wiped without an e2e run. Best-effort: a cleanup failure (no
|
|
DB reachable) must not fail the run. The spend_tracking dir goes on sys.path
|
|
only for this import and is removed after, so a broader `pytest tests/` run is
|
|
not left with a mutated path."""
|
|
if not session.stash.get(_E2E_TEST_RAN, False):
|
|
return
|
|
spend_dir = str(Path(__file__).parent / "spend_tracking")
|
|
sys.path.insert(0, spend_dir)
|
|
try:
|
|
from spend_e2e_client import reset_spend_logs # pyright: ignore
|
|
|
|
reset_spend_logs()
|
|
except Exception as exc: # noqa: BLE001 - cleanup is best-effort
|
|
print(f"spend-log cleanup skipped: {exc}")
|
|
finally:
|
|
if spend_dir in sys.path:
|
|
sys.path.remove(spend_dir)
|
|
|
|
|
|
@pytest.fixture
|
|
def resources(client: GatewayProvider) -> Iterator[ResourceManager]:
|
|
"""init -> run -> teardown: create a manager, run the test, release resources.
|
|
Cleanup goes through the shared Gateway, whatever the suite's client adds."""
|
|
manager = ResourceManager(client=client.gateway)
|
|
manager.init()
|
|
yield manager
|
|
manager.teardown()
|
|
|
|
|
|
@pytest.fixture
|
|
def scoped_key(resources: ResourceManager) -> str:
|
|
"""A fresh all-models key per test, auto-deleted by the resources teardown."""
|
|
return resources.key()
|