mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Add @step labels to the HttpTransport methods, the poll and wait helpers and the boot helpers that did real IO without recording a step, so a test that reaches the proxy through them no longer reports an empty or gappy step timeline in the JUnit report.
323 lines
16 KiB
Python
323 lines
16 KiB
Python
"""Generic configuration for live e2e tests against a running LiteLLM proxy.
|
|
|
|
Shared by every e2e suite under tests/e2e/. Values come from the
|
|
environment so the same tests run against localhost or a deployed proxy.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import socket
|
|
from dataclasses import dataclass
|
|
import time
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Final
|
|
|
|
from dotenv import load_dotenv
|
|
from e2e_metadata import step
|
|
from fixture_mode import deterministic_marker, parse_fixture_mode, registration_owner
|
|
from provider_edge import provider_edge_api_base
|
|
from pydantic import TypeAdapter
|
|
|
|
# Local runs keep provider / DataDog keys in tests/e2e/.env (see CONTRIBUTING.md).
|
|
# Compose injects them into the proxy container, but pytest on the host does not
|
|
# inherit that file unless we load it. override=False so a real shell export wins.
|
|
load_dotenv(Path(__file__).resolve().parent / ".env", override=False)
|
|
|
|
PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/")
|
|
MASTER_KEY = os.environ["LITELLM_MASTER_KEY"]
|
|
|
|
# Control-plane (management/admin) base URL. Defaults to PROXY_BASE_URL so a
|
|
# single path-routing host (stage ALB, compose monolith) works for both planes.
|
|
# Set LITELLM_CONTROL_PLANE_URL only when management is a different base than
|
|
# the LLM host and you are not going through an ingress that path-routes.
|
|
CONTROL_PLANE_BASE_URL = os.environ.get(
|
|
"LITELLM_CONTROL_PLANE_URL", PROXY_BASE_URL
|
|
).rstrip("/")
|
|
|
|
|
|
def split_replica_urls(raw: str) -> tuple[str, ...]:
|
|
return tuple(dict.fromkeys(url.strip().rstrip("/") for url in raw.split(",") if url.strip()))
|
|
|
|
|
|
def parse_replica_urls(raw: str, fallback: str) -> tuple[str, ...]:
|
|
return split_replica_urls(raw) or (fallback,)
|
|
|
|
|
|
def parse_control_plane_replica_urls(
|
|
raw: str, *, control_plane_base_url: str, base_url: str, replica_urls: tuple[str, ...]
|
|
) -> tuple[str, ...]:
|
|
"""The replicas a management read-back polls. LITELLM_CONTROL_PLANE_REPLICA_URLS
|
|
names them outright; unset, they follow the two base URLs: every data-plane
|
|
replica when the planes share a base (a monolith serves every route from every
|
|
replica) and the control-plane base alone when they differ. A stack sets it when
|
|
LITELLM_PROXY_REPLICA_URLS names gateway pods behind a shared router base, since
|
|
a gateway trims the management routes at startup and answers them 404."""
|
|
explicit: Final = split_replica_urls(raw)
|
|
if explicit:
|
|
return explicit
|
|
return replica_urls if control_plane_base_url == base_url else (control_plane_base_url,)
|
|
|
|
|
|
PROXY_REPLICA_URLS: Final = parse_replica_urls(os.environ.get("LITELLM_PROXY_REPLICA_URLS", ""), PROXY_BASE_URL)
|
|
CONTROL_PLANE_REPLICA_URLS: Final = parse_control_plane_replica_urls(
|
|
os.environ.get("LITELLM_CONTROL_PLANE_REPLICA_URLS", ""),
|
|
control_plane_base_url=CONTROL_PLANE_BASE_URL,
|
|
base_url=PROXY_BASE_URL,
|
|
replica_urls=PROXY_REPLICA_URLS,
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class StackEndpoints:
|
|
base_url: str
|
|
control_plane_base_url: str
|
|
replica_urls: tuple[str, ...]
|
|
control_replica_urls: tuple[str, ...]
|
|
|
|
def control_replica_urls_for(
|
|
self, *, base_url: str, control_plane_base_url: str, replica_urls: tuple[str, ...]
|
|
) -> tuple[str, ...]:
|
|
"""The control replicas a client built for these endpoints polls when its caller names none:
|
|
this stack's own list for this stack's endpoints, since an exported list describes one stack only,
|
|
and the base-URL rule for any other proxy."""
|
|
if (base_url, control_plane_base_url, replica_urls) == (
|
|
self.base_url,
|
|
self.control_plane_base_url,
|
|
self.replica_urls,
|
|
):
|
|
return self.control_replica_urls
|
|
return parse_control_plane_replica_urls(
|
|
"", control_plane_base_url=control_plane_base_url, base_url=base_url, replica_urls=replica_urls
|
|
)
|
|
|
|
|
|
ENV_STACK: Final = StackEndpoints(
|
|
base_url=PROXY_BASE_URL,
|
|
control_plane_base_url=CONTROL_PLANE_BASE_URL,
|
|
replica_urls=PROXY_REPLICA_URLS,
|
|
control_replica_urls=CONTROL_PLANE_REPLICA_URLS,
|
|
)
|
|
|
|
UI_USERNAME = os.environ.get("E2E_UI_USERNAME", "admin")
|
|
UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
|
|
|
|
# Dashboard base for playwright. Defaults to PROXY_BASE_URL so one ALB/monolith
|
|
# host covers /ui as well. Override E2E_UI_BASE_URL only if the UI is elsewhere.
|
|
UI_BASE_URL = os.environ.get("E2E_UI_BASE_URL", PROXY_BASE_URL).rstrip("/")
|
|
|
|
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
|
|
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
|
|
S3_PARTITION_GRANULARITY = os.environ.get("E2E_S3_PARTITION_GRANULARITY", "day")
|
|
|
|
LINEAR_MCP_URL = os.environ.get("E2E_LINEAR_MCP_URL", "https://mcp.linear.app/mcp")
|
|
LINEAR_STORAGE_STATE = os.environ.get("E2E_LINEAR_STORAGE_STATE", "")
|
|
LINEAR_READONLY_TOOL: Final = "list_teams" # as listed by tools/list on mcp.linear.app when PR #33787 landed
|
|
|
|
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
|
|
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
|
|
# read exported spans back through it.
|
|
OTEL_QUERY_URL = os.environ.get("E2E_OTEL_QUERY_URL", "http://localhost:16686").rstrip("/")
|
|
OTEL_EXPORTER_ENDPOINT = os.environ.get("E2E_OTEL_EXPORTER_ENDPOINT", "")
|
|
|
|
# Real-DataDog read-back (no local sink - destination fakes cannot be deployed
|
|
# on the cluster): the proxy delivers with DD_API_KEY as in production, and the
|
|
# tests read ingested events back through the DataDog Logs Search API, which
|
|
# additionally needs an application key. On the cluster the secret manager
|
|
# injects both; locally tests/e2e/.env provides them.
|
|
DD_SITE = os.environ.get("DD_SITE", "datadoghq.com").strip()
|
|
DD_API_KEY = os.environ.get("DD_API_KEY", "").strip()
|
|
DD_APP_KEY = os.environ.get("DD_APP_KEY", "").strip()
|
|
|
|
# After the first event is searchable, keep watching this long for a late
|
|
# duplicate before the exactly-one assertion: real-DataDog ingestion jitter can
|
|
# make one call's two events searchable tens of seconds apart, and a duplicate
|
|
# that surfaces late IS the bug (LIT-4447), so one poll interval is not enough.
|
|
DD_SETTLE_SECONDS = float(os.environ.get("E2E_DD_SETTLE_SECONDS", "30"))
|
|
# DataDog Logs Search `from` window (relative to now). Wide enough for a suite
|
|
# run plus ingestion lag; override if a long CI queue needs a wider lookback.
|
|
DD_SEARCH_FROM = os.environ.get("E2E_DD_SEARCH_FROM", "now-30m").strip() or "now-30m"
|
|
# The Logs Search API budget is tight - 2 requests per 10s org-wide
|
|
# (x-ratelimit-name logs_public_search_api) - so read-backs pace their search
|
|
# calls at this interval instead of POLL_INTERVAL, and back off when a 429
|
|
# still slips through (the budget is shared with anything else searching).
|
|
DD_SEARCH_INTERVAL = float(os.environ.get("E2E_DD_SEARCH_INTERVAL", "10"))
|
|
|
|
# Writes on the proxy are eventually consistent (e.g. spend rows flush on
|
|
# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once.
|
|
POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120"))
|
|
POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5"))
|
|
REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60"))
|
|
SLOW_PROVIDER_TIMEOUT_SECONDS = float(os.environ.get("E2E_SLOW_PROVIDER_TIMEOUT", "180"))
|
|
|
|
# How long a control-plane write (/model/new, /guardrails, /v1/agents) may take to
|
|
# reach EVERY replica. Distinct from POLL_TIMEOUT, which is sized for spend-row
|
|
# flush; this one is sized for the proxy's config reload
|
|
# (`proxy_config_reload_interval_seconds`, 30s by default and 7s on the e2e stack)
|
|
# plus margin.
|
|
#
|
|
# The barriers below wait this out on top of polling /v1/models on every replica in
|
|
# PROXY_REPLICA_URLS: that poll proves each addressed gateway converged, but not the
|
|
# workers behind it, and behind a load balancer (PROXY_REPLICA_URLS unset) a
|
|
# successful read only proves ONE replica converged, because every request opens a
|
|
# fresh connection and the next call re-rolls. See ProxyClient._await_model_servable.
|
|
PROPAGATION_TIMEOUT = float(os.environ.get("E2E_PROPAGATION_TIMEOUT", "15"))
|
|
|
|
# Record/replay fixture selection (see fixture_mode.py and provider_edge.py).
|
|
# The raw mode value is parsed and validated there; "live" (the default, also
|
|
# for empty values) means the harness behaves exactly as before this knob
|
|
# existed.
|
|
FIXTURE_MODE_RAW = os.environ.get("E2E_FIXTURE_MODE", "live")
|
|
FIXTURE_DIR = Path(
|
|
os.environ.get("E2E_FIXTURE_DIR", "").strip()
|
|
or str(Path(__file__).resolve().parent / ".fixtures")
|
|
)
|
|
|
|
# Where the provider-edge server binds, and the host name edge api_base URLs
|
|
# advertise to the proxy. They differ when the proxy runs in a container and
|
|
# reaches the pytest host via a gateway name like host.docker.internal.
|
|
PROVIDER_EDGE_BIND_HOST = os.environ.get("E2E_PROVIDER_EDGE_BIND_HOST", "").strip() or "127.0.0.1"
|
|
PROVIDER_EDGE_ADVERTISE_HOST = (
|
|
os.environ.get("E2E_PROVIDER_EDGE_ADVERTISE_HOST", "").strip() or PROVIDER_EDGE_BIND_HOST
|
|
)
|
|
|
|
# Deliberately modest concurrency. The suite shares its proxy with every other
|
|
# suite in the run, and 750 users at spawn rate 50 saturated the request path hard
|
|
# enough to distort latency-sensitive neighbours (and to spend real provider money
|
|
# fast).
|
|
LOAD_USERS = int(os.environ.get("E2E_LOAD_USERS", "200"))
|
|
LOAD_SPAWN_RATE = float(os.environ.get("E2E_LOAD_SPAWN_RATE", "20"))
|
|
LOAD_DURATION_SECONDS = float(os.environ.get("E2E_LOAD_DURATION_SECONDS", "60"))
|
|
LOAD_MAX_FAILURE_RATIO = float(os.environ.get("E2E_LOAD_MAX_FAILURE_RATIO", "0.01"))
|
|
|
|
# The throughput floor is derived per replica instead of being an absolute fleet
|
|
# number, so the verdict does not depend on how many replicas happen to be warm.
|
|
# One closed-loop user only ever occupies one replica at a time, so a short serial
|
|
# pass measures a single replica's request path: its throughput is 1/latency, and
|
|
# the concurrent phase then has to reach at least that much no matter how large
|
|
# the fleet is. An absolute floor instead asserted replicas x per-replica rate,
|
|
# which reactive autoscaling decides rather than the request path.
|
|
LOAD_BASELINE_SECONDS = float(os.environ.get("E2E_LOAD_BASELINE_SECONDS", "15"))
|
|
LOAD_MAX_SERIAL_LATENCY_SECONDS = float(os.environ.get("E2E_LOAD_MAX_SERIAL_LATENCY_SECONDS", "0.5"))
|
|
LOAD_MIN_CONCURRENCY_EFFICIENCY = float(os.environ.get("E2E_LOAD_MIN_CONCURRENCY_EFFICIENCY", "0.8"))
|
|
|
|
WEEKLY_ANOMALY_OPT_IN_ENV = "E2E_WEEKLY_ANOMALY"
|
|
MANAGED_FILES_OPT_IN_ENV = "E2E_MANAGED_FILES_STACK"
|
|
PROMPT_CACHING_OPT_IN_ENV = "E2E_PROMPT_CACHING_STACK"
|
|
REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS"
|
|
CLI_DETERMINISM_OPT_IN_ENV = "E2E_CLI_DETERMINISM"
|
|
MCP_OAUTH_LIVE_OPT_IN_ENV: Final = "E2E_MCP_OAUTH_LIVE"
|
|
PROVIDER_EDGE_HOST_OPT_IN_ENV: Final = "E2E_PROVIDER_EDGE_HOST_REACHABLE"
|
|
OWNED_GATEWAY_OPT_IN_ENV: Final = "E2E_OWNED_GATEWAY"
|
|
OTEL_V2_OPT_IN_ENV: Final = "E2E_OTEL_V2"
|
|
OTEL_TLS_OPT_IN_ENV: Final = "E2E_OTEL_EXPORTER_ENDPOINT"
|
|
SECRET_MANAGER_OPT_IN_ENV: Final = "E2E_SECRET_MANAGER"
|
|
ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6"))
|
|
ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6"))
|
|
ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3"))
|
|
ANOMALY_MAX_ERROR_RATIO = float(os.environ.get("E2E_ANOMALY_MAX_ERROR_RATIO", "0.05"))
|
|
ANOMALY_MIN_WARM_CACHE_READ_SHARE = float(
|
|
os.environ.get("E2E_ANOMALY_MIN_WARM_CACHE_READ_SHARE", "0.65")
|
|
)
|
|
ANOMALY_MAX_P95_TURN_SECONDS = float(
|
|
os.environ.get("E2E_ANOMALY_MAX_P95_TURN_SECONDS", "30")
|
|
)
|
|
ANOMALY_MAX_KEY_SPEND_USD = float(
|
|
os.environ.get("E2E_ANOMALY_MAX_KEY_SPEND_USD", "0.60")
|
|
)
|
|
ANOMALY_SPEND_SETTLE_SECONDS = float(
|
|
os.environ.get("E2E_ANOMALY_SPEND_SETTLE_SECONDS", "75")
|
|
)
|
|
MEMORY_REQUESTS_PER_PHASE = int(os.environ.get("E2E_MEMORY_REQUESTS_PER_PHASE", "300"))
|
|
MEMORY_RETRIES_PER_REQUEST = int(os.environ.get("E2E_MEMORY_RETRIES_PER_REQUEST", "2"))
|
|
MEMORY_TRANSCRIPT_TURNS = int(os.environ.get("E2E_MEMORY_TRANSCRIPT_TURNS", "40"))
|
|
MEMORY_CONCURRENCY = int(os.environ.get("E2E_MEMORY_CONCURRENCY", "4"))
|
|
MEMORY_RSS_SETTLE_SAMPLES = int(os.environ.get("E2E_MEMORY_RSS_SETTLE_SAMPLES", "15"))
|
|
MEMORY_RSS_SAMPLE_INTERVAL_SECONDS = float(os.environ.get("E2E_MEMORY_RSS_SAMPLE_INTERVAL_SECONDS", "1"))
|
|
MEMORY_RSS_BUDGET_MB = float(os.environ.get("E2E_MEMORY_RSS_BUDGET_MB", "48"))
|
|
MEMORY_IDLE_RSS_BUDGET_MB = float(os.environ.get("E2E_MEMORY_IDLE_RSS_BUDGET_MB", "768"))
|
|
MEMORY_STORED_REQUEST_BUDGET_KB = float(os.environ.get("E2E_MEMORY_STORED_REQUEST_BUDGET_KB", "64"))
|
|
|
|
|
|
def ws_base_url() -> str:
|
|
"""PROXY_BASE_URL with its scheme swapped for the websocket one, so a suite
|
|
opening a socket points at the same proxy every HTTP suite uses."""
|
|
for scheme, ws_scheme in (("https://", "wss://"), ("http://", "ws://")):
|
|
if PROXY_BASE_URL.startswith(scheme):
|
|
return ws_scheme + PROXY_BASE_URL[len(scheme) :]
|
|
return PROXY_BASE_URL
|
|
|
|
|
|
def datadog_mcp_url(*, toolsets: str = "core") -> str:
|
|
"""Regional Datadog remote MCP endpoint for this process's DD_SITE.
|
|
|
|
US1 is mcp.datadoghq.com; every other site is mcp.<site> (e.g. us5 ->
|
|
mcp.us5.datadoghq.com). A fixed mcp.datadoghq.com URL 403s when the keys
|
|
belong to a non-US1 org.
|
|
"""
|
|
site = (
|
|
os.environ.get("DD_SITE", DD_SITE) or "datadoghq.com"
|
|
).strip().removeprefix("https://").removeprefix("http://").rstrip("/")
|
|
site = site.removeprefix("app.")
|
|
host = "mcp.datadoghq.com" if site in ("", "datadoghq.com") else f"mcp.{site}"
|
|
base = f"https://{host}/v1/mcp"
|
|
return f"{base}?toolsets={toolsets}" if toolsets else base
|
|
|
|
|
|
def provider_edge_base(mount: str) -> str | None:
|
|
"""The api_base an edge-wired deployment should register with, using this
|
|
process's fixture-mode and edge-host configuration: None in live mode, the
|
|
shared edge server's mount URL in record and replay, and with the shared
|
|
cache on, the cache edge's mount URL scoped to the node that owns the
|
|
deployment: the running test, or the module or class whose fixture is
|
|
setting it up."""
|
|
return provider_edge_api_base(
|
|
mount,
|
|
mode_raw=FIXTURE_MODE_RAW,
|
|
bundle_dir=FIXTURE_DIR,
|
|
bind_host=PROVIDER_EDGE_BIND_HOST,
|
|
advertise_host=PROVIDER_EDGE_ADVERTISE_HOST,
|
|
test_key=registration_owner(),
|
|
forward_timeout=REQUEST_TIMEOUT,
|
|
)
|
|
|
|
|
|
STREAM_MIN_LEAD_SECONDS: Final = 1.0
|
|
|
|
|
|
def provider_paces_stream() -> bool:
|
|
return parse_fixture_mode(FIXTURE_MODE_RAW) != "replay"
|
|
|
|
|
|
def unique_marker() -> str:
|
|
"""A short unique token per call/run, so concurrent runs and the shared
|
|
response cache never collide on prompts, tags, or customer ids. In record
|
|
and replay modes the token is deterministic per test instead, so a replay
|
|
run regenerates the exact requests the record run sent."""
|
|
if parse_fixture_mode(FIXTURE_MODE_RAW) in ("record", "replay"):
|
|
return deterministic_marker()
|
|
return uuid.uuid4().hex[:12]
|
|
|
|
|
|
INHERITED_ENV_PREFIXES: Final = ("REDIS_", "MICROSOFT_", "GOOGLE_", "GENERIC_", "PROXY_")
|
|
|
|
|
|
def available_port() -> int:
|
|
with socket.socket() as listener:
|
|
listener.bind(("127.0.0.1", 0))
|
|
return TypeAdapter(tuple[str, int]).validate_python(listener.getsockname())[1]
|
|
|
|
|
|
@step("Wait for the last control-plane write to reach every proxy replica")
|
|
def settle_propagation(written_at: float) -> None:
|
|
"""Block until PROPAGATION_TIMEOUT has elapsed since `written_at`, a
|
|
`time.monotonic()` stamp taken the moment a control-plane write returned.
|
|
|
|
Callers that already polled for the object still need this: the poll proves one
|
|
replica has it, not all of them. Waiting out the config-reload budget is what
|
|
makes the object safe to use on whichever replica the next request lands on.
|
|
"""
|
|
remaining = PROPAGATION_TIMEOUT - (time.monotonic() - written_at)
|
|
if remaining > 0:
|
|
time.sleep(remaining)
|