litellm/tests/e2e/docker-compose.yml
devin-ai-integration[bot] fac43df9b9
fix(complexity_router): return empty dict from _classifier_call_metadata when metadata is absent (#33452)
* fix(complexity_router): return empty dict from _classifier_call_metadata when metadata is absent

The LLM classifier reads request_kwargs.get("litellm_metadata"), but the proxy stores request metadata under "metadata", so this returned None. _classifier_call_metadata then passed None straight through to the classifier acompletion call, which assumes a dict and blows up with 'NoneType' object has no attribute 'update'; the router swallowed it and silently fell back to heuristic scoring, so the configured LLM classifier never ran. Returning an empty dict keeps the classifier call well-formed.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(e2e): cover complexity-router LLM classifier routes over the proxy

Add a live e2e regression for the complexity auto-router: a lexically simple but hard prompt ("Is P equal to NP?") is routed by the LLM classifier to the higher-tier anthropic backend, read back from the spend log's model. Before the metadata fix the classifier silently crashed and the router fell back to heuristic SIMPLE scoring on the openai backend, so this test fails pre-fix and passes post-fix.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: Krrish Dholakia <krrishdholakia@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
2026-07-15 17:46:00 -07:00

157 lines
5.1 KiB
YAML

# local setup to run e2e tests
configs:
litellm_config:
content: |
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
database_url: os.environ/DATABASE_URL
store_prompts_in_spend_logs: true
proxy_budget_rescheduler_min_time: 5
proxy_budget_rescheduler_max_time: 10
litellm_settings:
drop_params: true
num_retries: 3
request_timeout: 600
cache: true
cache_params:
type: redis
host: redis
port: 6379
# OTEL v2 trace destination for the logging suite's trace-completeness
# tests: the arize_phoenix preset is OTLP with a configurable endpoint
# (PHOENIX_COLLECTOR_HTTP_ENDPOINT below points it at the jaeger service),
# so gen-AI spans export through a preset-owned provider - the code path
# where trace splits actually happen - with no cloud credentials needed.
callbacks: ["arize_phoenix"]
router_settings:
routing_strategy: simple-shuffle
num_retries: 3
allowed_fails: 5
cooldown_time: 30
fallbacks:
- gemini-2.5-flash: ["gpt-5.5", "claude-haiku-4-5"]
finetune_settings:
- custom_llm_provider: openai
api_key: os.environ/OPENAI_API_KEY
files_settings:
- custom_llm_provider: openai
api_key: os.environ/OPENAI_API_KEY
- custom_llm_provider: azure
api_base: os.environ/AZURE_API_BASE
api_key: os.environ/AZURE_API_KEY
api_version: "2024-05-01-preview"
model_list:
- model_name: gpt-5.5
litellm_params:
model: openai/gpt-5.5
api_key: os.environ/OPENAI_API_KEY
- model_name: claude-haiku-4-5
litellm_params:
model: anthropic/claude-haiku-4-5
api_key: os.environ/ANTHROPIC_API_KEY
- model_name: gemini-2.5-flash
litellm_params:
model: gemini/gemini-2.5-flash
api_key: os.environ/GEMINI_API_KEY
- model_name: openai-text-embedding-3-small
litellm_params:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY
# v2 auto-router with the LLM complexity classifier. SIMPLE stays on the
# openai backend; every higher tier routes to the anthropic backend, so the
# served deployment (read back from the spend log's model) reveals whether
# the LLM classifier actually ran or silently fell back to heuristic scoring.
- model_name: complexity-smart-router
litellm_params:
model: auto_router/complexity_router
complexity_router_config:
classifier_type: llm
classifier_llm_config:
model: gpt-5.5
tiers:
SIMPLE: gpt-5.5
MEDIUM: claude-haiku-4-5
COMPLEX: claude-haiku-4-5
REASONING: claude-haiku-4-5
services:
litellm:
image: ghcr.io/berriai/litellm:main-latest
depends_on:
db:
condition: service_healthy
redis:
condition: service_healthy
jaeger:
condition: service_healthy
env_file: .env
environment:
LITELLM_MASTER_KEY: sk-1234
STORE_MODEL_IN_DB: "True"
LITELLM_OTEL_V2: "true"
PHOENIX_COLLECTOR_HTTP_ENDPOINT: http://jaeger:4318/v1/traces
PHOENIX_API_KEY: local-jaeger-noauth
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
UI_USERNAME: admin
UI_PASSWORD: sk-1234
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME:-${AWS_BATCH_S3_BUCKET:-}}
AWS_BATCH_S3_BUCKET: ${AWS_BATCH_S3_BUCKET:-${AWS_S3_BUCKET_NAME:-}}
AWS_BATCH_ROLE_ARN: ${AWS_BATCH_ROLE_ARN:-}
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID:-}
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY:-}
AWS_REGION: ${AWS_REGION:-us-east-1}
GCS_BUCKET_NAME: ${GCS_BUCKET_NAME:-}
VERTEXAI_PROJECT: ${VERTEXAI_PROJECT:-}
VERTEXAI_CREDENTIALS: ${VERTEXAI_CREDENTIALS:-}
GOOGLE_APPLICATION_CREDENTIALS: ${GOOGLE_APPLICATION_CREDENTIALS:-}
MISTRAL_API_KEY: ${MISTRAL_API_KEY:-}
AZURE_API_BASE: ${AZURE_API_BASE:-}
AZURE_API_KEY: ${AZURE_API_KEY:-}
ports:
- "4000:4000"
configs:
- source: litellm_config
target: /app/config.yaml
command: ["--config", "/app/config.yaml", "--port", "4000"]
# throwaway db
db:
image: postgres:16
environment:
POSTGRES_USER: litellm
POSTGRES_PASSWORD: litellm
POSTGRES_DB: litellm
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm"]
interval: 3s
timeout: 3s
retries: 20
redis:
image: redis:7
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 3s
timeout: 3s
retries: 20
# throwaway OTEL trace destination (OTLP ingest on 4318 inside the network,
# query API on host 16686 for test read-back; see E2E_OTEL_QUERY_URL)
jaeger:
image: jaegertracing/all-in-one:1.62.0
ports:
- "16686:16686"
healthcheck:
test: ["CMD", "wget", "-qO-", "http://localhost:14269/"]
interval: 3s
timeout: 3s
retries: 20