litellm/tests/e2e/load/conftest.py
mubashir1osmani a150a3f85f fix(e2e): make load throughput suite use a fresh mock model and report failures
The stage load SLO was failing at ~95% errors while RPS looked fine. A fixed
load-mock name could reuse a stale deployment without mock_response, so Locust
hit real OpenAI. Always register a unique mock model, preflight one chat, hit
/v1/chat/completions, and attach a status/exception failure breakdown to the
assert so the next red run is diagnosable.
2026-07-18 14:08:18 -07:00

43 lines
1.4 KiB
Python

from __future__ import annotations
from collections.abc import Iterator
import pytest
from e2e_config import unique_marker
from load_client import LoadClient, build_client
from load_constants import LOAD_MOCK_PARAMS
from lifecycle import ResourceManager
from models import KeyGenerateBody
from proxy_client import ProxyClient
@pytest.fixture(scope="session")
def client(proxy: ProxyClient) -> LoadClient:
return build_client(proxy)
@pytest.fixture(scope="session")
def load_model(client: LoadClient) -> Iterator[str]:
"""Register a fresh mock deployment for this session and delete it after.
A fixed name like ``load-mock`` is unsafe on a shared stage proxy: a prior
run (or a hand-registered row) can leave a deployment without mock_response,
so Locust would hit real OpenAI with an invalid model and fail ~all requests
while the fixture skipped /model/new because the name was already listed.
"""
model_name = f"load-mock-{unique_marker()}"
model_id = client.proxy.create_model(model_name, LOAD_MOCK_PARAMS)
try:
yield model_name
finally:
client.proxy.delete_model(model_id)
@pytest.fixture
def load_key(resources: ResourceManager, client: LoadClient, load_model: str) -> str:
key = client.proxy.generate_key(
KeyGenerateBody(models=[load_model], user_id=f"e2e-load-{unique_marker()}")
)
resources.defer(lambda: client.proxy.delete_key(key))
return key