mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-06 08:16:43 +00:00
* fix(complexity_router): return empty dict from _classifier_call_metadata when metadata is absent
The LLM classifier reads request_kwargs.get("litellm_metadata"), but the proxy stores request metadata under "metadata", so this returned None. _classifier_call_metadata then passed None straight through to the classifier acompletion call, which assumes a dict and blows up with 'NoneType' object has no attribute 'update'; the router swallowed it and silently fell back to heuristic scoring, so the configured LLM classifier never ran. Returning an empty dict keeps the classifier call well-formed.
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* test(e2e): cover complexity-router LLM classifier routes over the proxy
Add a live e2e regression for the complexity auto-router: a lexically simple but hard prompt ("Is P equal to NP?") is routed by the LLM classifier to the higher-tier anthropic backend, read back from the spend log's model. Before the metadata fix the classifier silently crashed and the router fell back to heuristic SIMPLE scoring on the openai backend, so this test fails pre-fix and passes post-fix.
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---------
Co-authored-by: Krrish Dholakia <krrishdholakia@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
157 lines
5.1 KiB
YAML
157 lines
5.1 KiB
YAML
# local setup to run e2e tests
|
|
configs:
|
|
litellm_config:
|
|
content: |
|
|
general_settings:
|
|
master_key: os.environ/LITELLM_MASTER_KEY
|
|
database_url: os.environ/DATABASE_URL
|
|
store_prompts_in_spend_logs: true
|
|
proxy_budget_rescheduler_min_time: 5
|
|
proxy_budget_rescheduler_max_time: 10
|
|
|
|
litellm_settings:
|
|
drop_params: true
|
|
num_retries: 3
|
|
request_timeout: 600
|
|
cache: true
|
|
cache_params:
|
|
type: redis
|
|
host: redis
|
|
port: 6379
|
|
# OTEL v2 trace destination for the logging suite's trace-completeness
|
|
# tests: the arize_phoenix preset is OTLP with a configurable endpoint
|
|
# (PHOENIX_COLLECTOR_HTTP_ENDPOINT below points it at the jaeger service),
|
|
# so gen-AI spans export through a preset-owned provider - the code path
|
|
# where trace splits actually happen - with no cloud credentials needed.
|
|
callbacks: ["arize_phoenix"]
|
|
|
|
router_settings:
|
|
routing_strategy: simple-shuffle
|
|
num_retries: 3
|
|
allowed_fails: 5
|
|
cooldown_time: 30
|
|
fallbacks:
|
|
- gemini-2.5-flash: ["gpt-5.5", "claude-haiku-4-5"]
|
|
|
|
finetune_settings:
|
|
- custom_llm_provider: openai
|
|
api_key: os.environ/OPENAI_API_KEY
|
|
|
|
files_settings:
|
|
- custom_llm_provider: openai
|
|
api_key: os.environ/OPENAI_API_KEY
|
|
- custom_llm_provider: azure
|
|
api_base: os.environ/AZURE_API_BASE
|
|
api_key: os.environ/AZURE_API_KEY
|
|
api_version: "2024-05-01-preview"
|
|
|
|
model_list:
|
|
- model_name: gpt-5.5
|
|
litellm_params:
|
|
model: openai/gpt-5.5
|
|
api_key: os.environ/OPENAI_API_KEY
|
|
|
|
- model_name: claude-haiku-4-5
|
|
litellm_params:
|
|
model: anthropic/claude-haiku-4-5
|
|
api_key: os.environ/ANTHROPIC_API_KEY
|
|
|
|
- model_name: gemini-2.5-flash
|
|
litellm_params:
|
|
model: gemini/gemini-2.5-flash
|
|
api_key: os.environ/GEMINI_API_KEY
|
|
|
|
- model_name: openai-text-embedding-3-small
|
|
litellm_params:
|
|
model: openai/text-embedding-3-small
|
|
api_key: os.environ/OPENAI_API_KEY
|
|
|
|
# v2 auto-router with the LLM complexity classifier. SIMPLE stays on the
|
|
# openai backend; every higher tier routes to the anthropic backend, so the
|
|
# served deployment (read back from the spend log's model) reveals whether
|
|
# the LLM classifier actually ran or silently fell back to heuristic scoring.
|
|
- model_name: complexity-smart-router
|
|
litellm_params:
|
|
model: auto_router/complexity_router
|
|
complexity_router_config:
|
|
classifier_type: llm
|
|
classifier_llm_config:
|
|
model: gpt-5.5
|
|
tiers:
|
|
SIMPLE: gpt-5.5
|
|
MEDIUM: claude-haiku-4-5
|
|
COMPLEX: claude-haiku-4-5
|
|
REASONING: claude-haiku-4-5
|
|
|
|
services:
|
|
litellm:
|
|
image: ghcr.io/berriai/litellm:main-latest
|
|
depends_on:
|
|
db:
|
|
condition: service_healthy
|
|
redis:
|
|
condition: service_healthy
|
|
jaeger:
|
|
condition: service_healthy
|
|
env_file: .env
|
|
environment:
|
|
LITELLM_MASTER_KEY: sk-1234
|
|
STORE_MODEL_IN_DB: "True"
|
|
LITELLM_OTEL_V2: "true"
|
|
PHOENIX_COLLECTOR_HTTP_ENDPOINT: http://jaeger:4318/v1/traces
|
|
PHOENIX_API_KEY: local-jaeger-noauth
|
|
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
|
|
UI_USERNAME: admin
|
|
UI_PASSWORD: sk-1234
|
|
AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME:-${AWS_BATCH_S3_BUCKET:-}}
|
|
AWS_BATCH_S3_BUCKET: ${AWS_BATCH_S3_BUCKET:-${AWS_S3_BUCKET_NAME:-}}
|
|
AWS_BATCH_ROLE_ARN: ${AWS_BATCH_ROLE_ARN:-}
|
|
AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID:-}
|
|
AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY:-}
|
|
AWS_REGION: ${AWS_REGION:-us-east-1}
|
|
GCS_BUCKET_NAME: ${GCS_BUCKET_NAME:-}
|
|
VERTEXAI_PROJECT: ${VERTEXAI_PROJECT:-}
|
|
VERTEXAI_CREDENTIALS: ${VERTEXAI_CREDENTIALS:-}
|
|
GOOGLE_APPLICATION_CREDENTIALS: ${GOOGLE_APPLICATION_CREDENTIALS:-}
|
|
MISTRAL_API_KEY: ${MISTRAL_API_KEY:-}
|
|
AZURE_API_BASE: ${AZURE_API_BASE:-}
|
|
AZURE_API_KEY: ${AZURE_API_KEY:-}
|
|
ports:
|
|
- "4000:4000"
|
|
configs:
|
|
- source: litellm_config
|
|
target: /app/config.yaml
|
|
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
|
|
|
# throwaway db
|
|
db:
|
|
image: postgres:16
|
|
environment:
|
|
POSTGRES_USER: litellm
|
|
POSTGRES_PASSWORD: litellm
|
|
POSTGRES_DB: litellm
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U litellm"]
|
|
interval: 3s
|
|
timeout: 3s
|
|
retries: 20
|
|
|
|
redis:
|
|
image: redis:7
|
|
healthcheck:
|
|
test: ["CMD", "redis-cli", "ping"]
|
|
interval: 3s
|
|
timeout: 3s
|
|
retries: 20
|
|
|
|
# throwaway OTEL trace destination (OTLP ingest on 4318 inside the network,
|
|
# query API on host 16686 for test read-back; see E2E_OTEL_QUERY_URL)
|
|
jaeger:
|
|
image: jaegertracing/all-in-one:1.62.0
|
|
ports:
|
|
- "16686:16686"
|
|
healthcheck:
|
|
test: ["CMD", "wget", "-qO-", "http://localhost:14269/"]
|
|
interval: 3s
|
|
timeout: 3s
|
|
retries: 20
|