From 66dea7df8feb0a59d84e090d43ecf7fcb391ba95 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sat, 18 Jul 2026 12:50:23 -0700 Subject: [PATCH] chore(e2e): remove tests/e2e/docker-compose.yml (#33837) Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/e2e/CLAUDE.md | 2 +- tests/e2e/CONTRIBUTING.md | 38 ++++---- tests/e2e/docker-compose.yml | 165 ----------------------------------- 3 files changed, 19 insertions(+), 186 deletions(-) delete mode 100644 tests/e2e/docker-compose.yml diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 3130a4e1e16..1fa78275085 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -194,6 +194,6 @@ other... - when it comes to typing an input schema for an api endpoint, have it type X = A | B | C ... where X = exhaustive union of all supported input schemas and A, B, C typically are composed by a base type. types are only pretty for a api request / response body. make sure to compose types instead of repeating the same base attributes over and over again. -- use the docker-compose to your advantage and spin up a local proxy, make sure all tests pass. if a test fails due to an internally found issue, let users know to create a linear ticket for it. +- spin up a local proxy by running the litellm proxy locally (`litellm --config .yml --port 4000`; see CONTRIBUTING.md), make sure all tests pass. if a test fails due to an internally found issue, let users know to create a linear ticket for it. - do not use xfail markers, tests should be written in a form that the end user expects it to pass diff --git a/tests/e2e/CONTRIBUTING.md b/tests/e2e/CONTRIBUTING.md index 49d776cd64b..fc43769aca8 100644 --- a/tests/e2e/CONTRIBUTING.md +++ b/tests/e2e/CONTRIBUTING.md @@ -9,26 +9,33 @@ When contributing to this directory, please first discuss the change you wish to ## Setup -The suites run against a live proxy, so bring one up first. `docker-compose.yml` here starts that proxy with a throwaway Postgres and Redis; `docker compose down -v` resets everything, so no state leaks between runs. The proxy config is inlined in the compose file under `configs`, prewired with example models (`gpt-5.5`, `claude-haiku-4-5`, `gemini-2.5-flash`, `openai-text-embedding-3-small`) whose keys come from your `.env`. If your test needs another model, a pricing override, or a guardrail declared up front, add it to that inline config and read it back in the test rather than hardcoding values +The suites run against a live proxy, so bring one up first by running the litellm proxy locally. Point it at a config that prewires the example models the suites use (`gpt-5.5`, `claude-haiku-4-5`, `gemini-2.5-flash`, `openai-text-embedding-3-small`) with keys from your `.env`, and enables prompt storage, a redis cache, and the fast budget rescheduler the quota suites rely on. If your test needs another model, a pricing override, or a guardrail declared up front, add it to that config and read it back in the test rather than hardcoding values ## Running the tests locally -1. Create a `.env` file in this directory with the provider keys the example models use: +1. Create a `.env` file in this directory with the provider keys the example models use, plus the master key and the Postgres/Redis coordinates your config reads back: ```bash + LITELLM_MASTER_KEY="sk-1234" + DATABASE_URL="postgresql://llmproxy:dbpassword9090@localhost:5432/litellm" + REDIS_HOST="localhost" + REDIS_PORT="6379" OPENAI_API_KEY="sk-..." ANTHROPIC_API_KEY="sk-..." GEMINI_API_KEY="..." ``` -2. Bring the stack up from this directory: +2. Bring up a Postgres and a Redis for the proxy to use. The repo-root `docker-compose.yml` already defines a Postgres on `5432`; a `docker run -p 6379:6379 redis:7` covers Redis. Point `DATABASE_URL` / `REDIS_HOST` / `REDIS_PORT` at whatever you run + +3. Start the litellm proxy locally against your config and confirm it is live: ```bash - docker compose up -d + set -a && source .env && set +a + litellm --config .yml --port 4000 curl -fs http://localhost:4000/health/liveliness ``` -3. Run a suite against it; the harness reads `LITELLM_PROXY_URL` (default `http://localhost:4000`): +4. Run a suite against it; the harness reads `LITELLM_PROXY_URL` (default `http://localhost:4000`): ```bash uv run pytest tests/e2e/llm_translation/ -v @@ -41,20 +48,11 @@ The suites run against a live proxy, so bring one up first. `docker-compose.yml` uv run playwright install chromium ``` - They also need a proxy whose bundled UI contains the change under test. The published `main-latest` image ships the UI from the last release; to test local UI changes, build the image from your branch and point the compose stack at it: + They also need a proxy whose bundled UI contains the change under test, so run the proxy from your branch (an editable install serves the UI your checkout builds) - ```bash - docker build -t litellm-local . - LITELLM_E2E_IMAGE=litellm-local docker compose up -d - ``` +Some suites need extra services the bare proxy does not start. The `logging/` OTEL trace-completeness tests read spans back from a jaeger query API at `http://localhost:16686` (override with `E2E_OTEL_QUERY_URL`); run a `jaegertracing/all-in-one` and point `PHOENIX_COLLECTOR_HTTP_ENDPOINT` at its OTLP ingest. The `mcp/` suite needs the deterministic upstream MCP server in `mcp_tests/mcp_e2e_upstream_server.py` reachable by the proxy -4. Tear it down when you're done: - - ```bash - docker compose down -v - ``` - -Tests marked `@pytest.mark.e2e` hard-fail when no proxy answers `/health/liveliness`, so a run that goes red with `No live proxy` at setup means the stack isn't up; they never skip for a missing proxy, so an absent stack can't be mistaken for a pass +Tests marked `@pytest.mark.e2e` hard-fail when no proxy answers `/health/liveliness`, so a run that goes red with `No live proxy` at setup means the proxy isn't up; they never skip for a missing proxy, so an absent proxy can't be mistaken for a pass ## What a complete test looks like @@ -142,12 +140,12 @@ Before you push 1. Run `make lint-e2e-basedpyright` (or `make pre-commit` with your changes staged); the harness is fully typed and the gate allows zero basedpyright errors, enforced in CI on any PR touching `tests/e2e/**/*.py` -2. Add the models your test needs to the inline config in `docker-compose.yml` +2. Add the models your test needs to the config your local proxy loads -3. Bring the stack up and run your suite against it: +3. Start the litellm proxy locally and run your suite against it: ```bash - docker compose up -d + litellm --config .yml --port 4000 uv run pytest tests/e2e// -v ``` diff --git a/tests/e2e/docker-compose.yml b/tests/e2e/docker-compose.yml deleted file mode 100644 index c1ce8eccc3e..00000000000 --- a/tests/e2e/docker-compose.yml +++ /dev/null @@ -1,165 +0,0 @@ -# local setup to run e2e tests -configs: - litellm_config: - content: | - general_settings: - master_key: os.environ/LITELLM_MASTER_KEY - database_url: os.environ/DATABASE_URL - store_prompts_in_spend_logs: true - proxy_budget_rescheduler_min_time: 5 - proxy_budget_rescheduler_max_time: 10 - - litellm_settings: - drop_params: true - num_retries: 3 - request_timeout: 600 - cache: true - cache_params: - type: redis - host: redis - port: 6379 - # OTEL v2 trace destination for the logging suite's trace-completeness - # tests: the arize_phoenix preset is OTLP with a configurable endpoint - # (PHOENIX_COLLECTOR_HTTP_ENDPOINT below points it at the jaeger service), - # so gen-AI spans export through a preset-owned provider - the code path - # where trace splits actually happen - with no cloud credentials needed. - callbacks: ["arize_phoenix", "datadog"] - - router_settings: - routing_strategy: simple-shuffle - num_retries: 3 - allowed_fails: 5 - cooldown_time: 30 - fallbacks: - - gemini-2.5-flash: ["gpt-5.5", "claude-haiku-4-5"] - - finetune_settings: - - custom_llm_provider: openai - api_key: os.environ/OPENAI_API_KEY - - files_settings: - - custom_llm_provider: openai - api_key: os.environ/OPENAI_API_KEY - - custom_llm_provider: azure - api_base: os.environ/AZURE_API_BASE - api_key: os.environ/AZURE_API_KEY - api_version: "2024-05-01-preview" - - model_list: - - model_name: gpt-5.5 - litellm_params: - model: openai/gpt-5.5 - api_key: os.environ/OPENAI_API_KEY - - - model_name: claude-haiku-4-5 - litellm_params: - model: anthropic/claude-haiku-4-5 - api_key: os.environ/ANTHROPIC_API_KEY - - - model_name: gemini-2.5-flash - litellm_params: - model: gemini/gemini-2.5-flash - api_key: os.environ/GEMINI_API_KEY - - - model_name: openai-text-embedding-3-small - litellm_params: - model: openai/text-embedding-3-small - api_key: os.environ/OPENAI_API_KEY - - # v2 auto-router with the LLM complexity classifier. SIMPLE stays on the - # openai backend; every higher tier routes to the anthropic backend, so the - # served deployment (read back from the spend log's model) reveals whether - # the LLM classifier actually ran or silently fell back to heuristic scoring. - - model_name: complexity-smart-router - litellm_params: - model: auto_router/complexity_router - complexity_router_config: - classifier_type: llm - classifier_llm_config: - model: gpt-5.5 - tiers: - SIMPLE: gpt-5.5 - MEDIUM: claude-haiku-4-5 - COMPLEX: claude-haiku-4-5 - REASONING: claude-haiku-4-5 - -services: - litellm: - image: ghcr.io/berriai/litellm:main-latest - depends_on: - db: - condition: service_healthy - redis: - condition: service_healthy - jaeger: - condition: service_healthy - env_file: .env - environment: - LITELLM_MASTER_KEY: sk-1234 - STORE_MODEL_IN_DB: "True" - # Real DataDog delivery (no local sink): the key comes from the - # environment - the cluster's secret manager injects it, locally - # tests/e2e/.env provides it. Tests read delivery back via the DataDog - # Logs Search API (DD_APP_KEY, test-side only - see logging/datadog_reader.py). - DD_API_KEY: ${DD_API_KEY:-} - DD_SITE: ${DD_SITE:-datadoghq.com} - LITELLM_OTEL_V2: "true" - PHOENIX_COLLECTOR_HTTP_ENDPOINT: http://jaeger:4318/v1/traces - PHOENIX_API_KEY: local-jaeger-noauth - DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm - UI_USERNAME: admin - UI_PASSWORD: sk-1234 - AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME:-${AWS_BATCH_S3_BUCKET:-}} - AWS_BATCH_S3_BUCKET: ${AWS_BATCH_S3_BUCKET:-${AWS_S3_BUCKET_NAME:-}} - AWS_BATCH_ROLE_ARN: ${AWS_BATCH_ROLE_ARN:-} - AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID:-} - AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY:-} - AWS_REGION: ${AWS_REGION:-us-east-1} - GCS_BUCKET_NAME: ${GCS_BUCKET_NAME:-} - VERTEXAI_PROJECT: ${VERTEXAI_PROJECT:-} - VERTEXAI_CREDENTIALS: ${VERTEXAI_CREDENTIALS:-} - GOOGLE_APPLICATION_CREDENTIALS: ${GOOGLE_APPLICATION_CREDENTIALS:-} - MISTRAL_API_KEY: ${MISTRAL_API_KEY:-} - AZURE_API_BASE: ${AZURE_API_BASE:-} - AZURE_API_KEY: ${AZURE_API_KEY:-} - AZURE_AI_API_BASE: ${AZURE_AI_API_BASE:-} - AZURE_AI_API_KEY: ${AZURE_AI_API_KEY:-} - ports: - - "4000:4000" - configs: - - source: litellm_config - target: /app/config.yaml - command: ["--config", "/app/config.yaml", "--port", "4000"] - -# throwaway db - db: - image: postgres:16 - environment: - POSTGRES_USER: litellm - POSTGRES_PASSWORD: litellm - POSTGRES_DB: litellm - healthcheck: - test: ["CMD-SHELL", "pg_isready -U litellm"] - interval: 3s - timeout: 3s - retries: 20 - - redis: - image: redis:7 - healthcheck: - test: ["CMD", "redis-cli", "ping"] - interval: 3s - timeout: 3s - retries: 20 - -# throwaway OTEL trace destination (OTLP ingest on 4318 inside the network, -# query API on host 16686 for test read-back; see E2E_OTEL_QUERY_URL) - jaeger: - image: jaegertracing/all-in-one:1.62.0 - ports: - - "16686:16686" - healthcheck: - test: ["CMD", "wget", "-qO-", "http://localhost:14269/"] - interval: 3s - timeout: 3s - retries: 20