litellm/tests/e2e/e2e_config.py
mateo-berri 187cab5049 test(e2e): guard Azure reasoning_effort=none via base_model and xfail the dual token-param bug
Adds two customer-regression rows to the Azure chat e2e suite. The first
sends reasoning_effort=none to a custom-named deployment (gpt-5.6-sol-e2e,
swappable via E2E_AZURE_CUSTOM_MODEL) whose capabilities resolve through
base_model, asserting the request completes with zero reasoning tokens on
a prompt that reasons at default effort, so both a gate 400 (GH #31243,
SDK fix in PR #28490) and a silently dropped param fail the row. The
second, a strict xfail until GH #31614 is fixed, sends a client
max_completion_tokens to a gpt-4o deployment carrying a config-level
max_tokens default; the proxy forwards both and Azure rejects the pair
2026-07-16 15:41:00 -07:00

60 lines
2.5 KiB
Python

"""Generic configuration for live e2e tests against a running LiteLLM proxy.
Shared by every e2e suite under tests/e2e/. Values come from the
environment so the same tests run against localhost or a deployed proxy.
"""
import os
import uuid
PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/")
MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234")
# Control-plane (management/admin) base URL. In a split control-plane/data-plane
# deployment the LLM data plane (PROXY_BASE_URL: /chat, /embeddings, native
# passthrough) and the management API (keys, users, teams, orgs, budgets, spend,
# model info, /openapi.json) are served by *different* services. The suite drives
# both through one Transport that routes by path (see transport.SplitTransport).
# Defaults to PROXY_BASE_URL so a monolithic proxy serving everything on one URL
# behaves exactly as before.
CONTROL_PLANE_BASE_URL = os.environ.get(
"LITELLM_CONTROL_PLANE_URL", PROXY_BASE_URL
).rstrip("/")
UI_USERNAME = os.environ.get("E2E_UI_USERNAME", "admin")
UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
AZURE_CHAT_MODELS = tuple(
f"azure-{os.environ.get(var, default)}"
for var, default in (
("E2E_AZURE_SOL_MODEL", "gpt-5.6-sol"),
("E2E_AZURE_TERRA_MODEL", "gpt-5.6-terra"),
("E2E_AZURE_LUNA_MODEL", "gpt-5.6-luna"),
)
)
AZURE_CUSTOM_NAME_CHAT_MODEL = (
f"azure-{os.environ.get('E2E_AZURE_CUSTOM_MODEL', 'gpt-5.6-sol-e2e')}"
)
AZURE_GPT4O_CHAT_MODEL = f"azure-{os.environ.get('E2E_AZURE_GPT4O_MODEL', 'gpt-4o')}"
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
# read exported spans back through it.
OTEL_QUERY_URL = os.environ.get("E2E_OTEL_QUERY_URL", "http://localhost:16686").rstrip("/")
# Writes on the proxy are eventually consistent (e.g. spend rows flush on
# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once.
POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120"))
POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5"))
REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60"))
def unique_marker() -> str:
"""A short unique token per call/run, so concurrent runs and the shared
response cache never collide on prompts, tags, or customer ids."""
return uuid.uuid4().hex[:12]