mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-13 23:11:40 +00:00
* test(e2e): harden stage flakes for batches, UI, and MCP Unique batch model names avoid load-balancing onto stale azure-batch deployments that still pointed at the retired gpt-4.1-mini-batch, which only the managed/unified path was hitting. Retry batch retrieve on 500 and /ui/api-keys navigation on ERR_ABORTED. Skip the MCP key-access suite when the compose-only mcp-upstream is unreachable on stage k8s * test(e2e): cover Datadog remote MCP via search_datadog_logs Register the regional Datadog MCP endpoint with DD-API-KEY / DD-APPLICATION-KEY static headers (CI-safe header auth; browser OAuth is not headless-automatable). Seed a chat completion marked e2e-datadog-mcp-*, assert the proxy shipped it, list tools, call search_datadog_logs for the marker, and delete the server on teardown. Math-upstream key-access tests only skip when that compose service is unreachable * test(e2e): drop compose math MCP upstream; use Datadog only Key-access denial and happy-path MCP e2e both register the real regional Datadog remote MCP server with DD-API-KEY / DD-APPLICATION-KEY headers. Remove the mcp-upstream compose service and FastMCP add/multiply fixture * docs(e2e): require real Datadog MCP for all mcp suite tests Document that tests/e2e/mcp must register via datadog_mcp helpers against mcp.<site>/v1/mcp and must not introduce compose or fake MCP upstreams * chore: restore mcp_e2e_upstream_server.py Keep the FastMCP fixture file; e2e no longer wires it in compose, but the module itself is not part of the Datadog-only cleanup * fix(e2e): load tests/e2e/.env and fix datadog_reader importlib load pytest on the host never inherited compose env_file keys, so DD_API_KEY stayed empty. load_dotenv tests/e2e/.env in e2e_config. Register the dynamically loaded datadog_reader module in sys.modules so dataclasses do not crash under Python 3.12 * test(e2e/batches): harden azure/vertex unified lifecycle flakes Put the provider deployment name in every JSONL body so Azure does not depend on a perfect model rewrite. Retry create/retrieve/cancel on transient statuses with backoff. Drop cancel assertions for azure and vertex (registry only has a shared basic cell; create+retrieve prove routing, cancel stays best-effort cleanup) * test(e2e/ui): treat api-keys shell as success after SPA ERR_ABORTED Post-login client redirects abort the first /ui/api-keys/ goto on stage. Wait off /ui/login after cookie set, then accept the page once Create New Key is visible even if goto raised ERR_ABORTED * test(e2e): drop flaky key models dropdown Playwright suite API management e2e already covers key generate/update persistence. The UI Models-dropdown sentinel cases only added SPA ERR_ABORTED noise and no unique product signal. Remove the suite and unused browser fixtures
102 lines
5.1 KiB
Python
102 lines
5.1 KiB
Python
"""Generic configuration for live e2e tests against a running LiteLLM proxy.
|
|
|
|
Shared by every e2e suite under tests/e2e/. Values come from the
|
|
environment so the same tests run against localhost or a deployed proxy.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import uuid
|
|
from pathlib import Path
|
|
|
|
from dotenv import load_dotenv
|
|
|
|
# Local runs keep provider / DataDog keys in tests/e2e/.env (see CONTRIBUTING.md).
|
|
# Compose injects them into the proxy container, but pytest on the host does not
|
|
# inherit that file unless we load it. override=False so a real shell export wins.
|
|
load_dotenv(Path(__file__).resolve().parent / ".env", override=False)
|
|
|
|
PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/")
|
|
MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234")
|
|
|
|
# Control-plane (management/admin) base URL. Defaults to PROXY_BASE_URL so a
|
|
# single path-routing host (stage ALB, compose monolith) works for both planes.
|
|
# Set LITELLM_CONTROL_PLANE_URL only when management is a different base than
|
|
# the LLM host and you are not going through an ingress that path-routes.
|
|
CONTROL_PLANE_BASE_URL = os.environ.get(
|
|
"LITELLM_CONTROL_PLANE_URL", PROXY_BASE_URL
|
|
).rstrip("/")
|
|
|
|
UI_USERNAME = os.environ.get("E2E_UI_USERNAME", "admin")
|
|
UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
|
|
|
|
# Dashboard base for playwright. Defaults to PROXY_BASE_URL so one ALB/monolith
|
|
# host covers /ui as well. Override E2E_UI_BASE_URL only if the UI is elsewhere.
|
|
UI_BASE_URL = os.environ.get("E2E_UI_BASE_URL", PROXY_BASE_URL).rstrip("/")
|
|
|
|
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
|
|
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
|
|
|
|
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
|
|
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
|
|
# read exported spans back through it.
|
|
OTEL_QUERY_URL = os.environ.get("E2E_OTEL_QUERY_URL", "http://localhost:16686").rstrip("/")
|
|
|
|
# Real-DataDog read-back (no local sink - destination fakes cannot be deployed
|
|
# on the cluster): the proxy delivers with DD_API_KEY as in production, and the
|
|
# tests read ingested events back through the DataDog Logs Search API, which
|
|
# additionally needs an application key. On the cluster the secret manager
|
|
# injects both; locally tests/e2e/.env provides them.
|
|
DD_SITE = os.environ.get("DD_SITE", "datadoghq.com").strip()
|
|
DD_API_KEY = os.environ.get("DD_API_KEY", "").strip()
|
|
DD_APP_KEY = os.environ.get("DD_APP_KEY", "").strip()
|
|
|
|
# After the first event is searchable, keep watching this long for a late
|
|
# duplicate before the exactly-one assertion: real-DataDog ingestion jitter can
|
|
# make one call's two events searchable tens of seconds apart, and a duplicate
|
|
# that surfaces late IS the bug (LIT-4447), so one poll interval is not enough.
|
|
DD_SETTLE_SECONDS = float(os.environ.get("E2E_DD_SETTLE_SECONDS", "30"))
|
|
# DataDog Logs Search `from` window (relative to now). Wide enough for a suite
|
|
# run plus ingestion lag; override if a long CI queue needs a wider lookback.
|
|
DD_SEARCH_FROM = os.environ.get("E2E_DD_SEARCH_FROM", "now-30m").strip() or "now-30m"
|
|
# The Logs Search API budget is tight - 2 requests per 10s org-wide
|
|
# (x-ratelimit-name logs_public_search_api) - so read-backs pace their search
|
|
# calls at this interval instead of POLL_INTERVAL, and back off when a 429
|
|
# still slips through (the budget is shared with anything else searching).
|
|
DD_SEARCH_INTERVAL = float(os.environ.get("E2E_DD_SEARCH_INTERVAL", "10"))
|
|
|
|
# Writes on the proxy are eventually consistent (e.g. spend rows flush on
|
|
# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once.
|
|
POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120"))
|
|
POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5"))
|
|
REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60"))
|
|
|
|
LOAD_USERS = int(os.environ.get("E2E_LOAD_USERS", "750"))
|
|
LOAD_SPAWN_RATE = float(os.environ.get("E2E_LOAD_SPAWN_RATE", "50"))
|
|
LOAD_DURATION_SECONDS = float(os.environ.get("E2E_LOAD_DURATION_SECONDS", "60"))
|
|
LOAD_MIN_RPS = float(os.environ.get("E2E_LOAD_MIN_RPS", "355"))
|
|
LOAD_MAX_FAILURE_RATIO = float(os.environ.get("E2E_LOAD_MAX_FAILURE_RATIO", "0.01"))
|
|
|
|
|
|
def datadog_mcp_url(*, toolsets: str = "core") -> str:
|
|
"""Regional Datadog remote MCP endpoint for this process's DD_SITE.
|
|
|
|
US1 is mcp.datadoghq.com; every other site is mcp.<site> (e.g. us5 ->
|
|
mcp.us5.datadoghq.com). A fixed mcp.datadoghq.com URL 403s when the keys
|
|
belong to a non-US1 org.
|
|
"""
|
|
site = (
|
|
os.environ.get("DD_SITE", DD_SITE) or "datadoghq.com"
|
|
).strip().removeprefix("https://").removeprefix("http://").rstrip("/")
|
|
if site.startswith("app."):
|
|
site = site[len("app.") :]
|
|
host = "mcp.datadoghq.com" if site in ("", "datadoghq.com") else f"mcp.{site}"
|
|
base = f"https://{host}/v1/mcp"
|
|
return f"{base}?toolsets={toolsets}" if toolsets else base
|
|
|
|
|
|
def unique_marker() -> str:
|
|
"""A short unique token per call/run, so concurrent runs and the shared
|
|
response cache never collide on prompts, tags, or customer ids."""
|
|
return uuid.uuid4().hex[:12]
|