mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
Adds two customer-regression rows to the Azure chat e2e suite. The first sends reasoning_effort=none to a custom-named deployment (gpt-5.6-sol-e2e, swappable via E2E_AZURE_CUSTOM_MODEL) whose capabilities resolve through base_model, asserting the request completes with zero reasoning tokens on a prompt that reasons at default effort, so both a gate 400 (GH #31243, SDK fix in PR #28490) and a silently dropped param fail the row. The second, a strict xfail until GH #31614 is fixed, sends a client max_completion_tokens to a gpt-4o deployment carrying a config-level max_tokens default; the proxy forwards both and Azure rejects the pair
60 lines
2.5 KiB
Python
60 lines
2.5 KiB
Python
"""Generic configuration for live e2e tests against a running LiteLLM proxy.
|
|
|
|
Shared by every e2e suite under tests/e2e/. Values come from the
|
|
environment so the same tests run against localhost or a deployed proxy.
|
|
"""
|
|
|
|
import os
|
|
import uuid
|
|
|
|
PROXY_BASE_URL = os.environ.get("LITELLM_PROXY_URL", "http://localhost:4000").rstrip("/")
|
|
MASTER_KEY = os.environ.get("LITELLM_MASTER_KEY", "sk-1234")
|
|
|
|
# Control-plane (management/admin) base URL. In a split control-plane/data-plane
|
|
# deployment the LLM data plane (PROXY_BASE_URL: /chat, /embeddings, native
|
|
# passthrough) and the management API (keys, users, teams, orgs, budgets, spend,
|
|
# model info, /openapi.json) are served by *different* services. The suite drives
|
|
# both through one Transport that routes by path (see transport.SplitTransport).
|
|
# Defaults to PROXY_BASE_URL so a monolithic proxy serving everything on one URL
|
|
# behaves exactly as before.
|
|
CONTROL_PLANE_BASE_URL = os.environ.get(
|
|
"LITELLM_CONTROL_PLANE_URL", PROXY_BASE_URL
|
|
).rstrip("/")
|
|
|
|
UI_USERNAME = os.environ.get("E2E_UI_USERNAME", "admin")
|
|
UI_PASSWORD = os.environ.get("E2E_UI_PASSWORD", MASTER_KEY)
|
|
|
|
CHEAP_ANTHROPIC_MODEL = os.environ.get("E2E_CHEAP_ANTHROPIC_MODEL", "claude-haiku-4-5")
|
|
CHEAP_OPENAI_MODEL = os.environ.get("E2E_CHEAP_OPENAI_MODEL", "gpt-5.5")
|
|
|
|
AZURE_CHAT_MODELS = tuple(
|
|
f"azure-{os.environ.get(var, default)}"
|
|
for var, default in (
|
|
("E2E_AZURE_SOL_MODEL", "gpt-5.6-sol"),
|
|
("E2E_AZURE_TERRA_MODEL", "gpt-5.6-terra"),
|
|
("E2E_AZURE_LUNA_MODEL", "gpt-5.6-luna"),
|
|
)
|
|
)
|
|
|
|
AZURE_CUSTOM_NAME_CHAT_MODEL = (
|
|
f"azure-{os.environ.get('E2E_AZURE_CUSTOM_MODEL', 'gpt-5.6-sol-e2e')}"
|
|
)
|
|
|
|
AZURE_GPT4O_CHAT_MODEL = f"azure-{os.environ.get('E2E_AZURE_GPT4O_MODEL', 'gpt-4o')}"
|
|
|
|
# Jaeger query API of the compose stack's OTEL trace destination (the `jaeger`
|
|
# service in docker-compose.yml maps it to host 16686). Trace-completeness tests
|
|
# read exported spans back through it.
|
|
OTEL_QUERY_URL = os.environ.get("E2E_OTEL_QUERY_URL", "http://localhost:16686").rstrip("/")
|
|
|
|
# Writes on the proxy are eventually consistent (e.g. spend rows flush on
|
|
# proxy_batch_write_at, ~60s). Read-backs poll to this deadline, never sleep-once.
|
|
POLL_TIMEOUT = float(os.environ.get("E2E_POLL_TIMEOUT", "120"))
|
|
POLL_INTERVAL = float(os.environ.get("E2E_POLL_INTERVAL", "5"))
|
|
REQUEST_TIMEOUT = float(os.environ.get("E2E_REQUEST_TIMEOUT", "60"))
|
|
|
|
|
|
def unique_marker() -> str:
|
|
"""A short unique token per call/run, so concurrent runs and the shared
|
|
response cache never collide on prompts, tags, or customer ids."""
|
|
return uuid.uuid4().hex[:12]
|