# local setup to run e2e tests configs: dd_sink_script: content: | # Minimal DataDog logs-intake sink for the logging suite: records every # POST (gunzipping the compressed batches the integration sends) and # replays them as JSON on GET /requests so tests can assert delivery. import gzip, json from http.server import BaseHTTPRequestHandler, HTTPServer REQUESTS = [] class Handler(BaseHTTPRequestHandler): def do_POST(self): body = self.rfile.read(int(self.headers.get("Content-Length", 0))) if self.headers.get("Content-Encoding") == "gzip": body = gzip.decompress(body) REQUESTS.append({"path": self.path, "body": body.decode("utf-8", "replace")}) self.send_response(202) self.end_headers() self.wfile.write(b"{}") def do_GET(self): self.send_response(200) if self.path == "/health": self.send_header("Content-Type", "text/plain") self.end_headers() self.wfile.write(b"ok") return self.send_header("Content-Type", "application/json") self.end_headers() self.wfile.write(json.dumps({"requests": REQUESTS}).encode()) def log_message(self, *args): pass HTTPServer(("0.0.0.0", 8080), Handler).serve_forever() litellm_config: content: | general_settings: master_key: os.environ/LITELLM_MASTER_KEY database_url: os.environ/DATABASE_URL store_prompts_in_spend_logs: true proxy_budget_rescheduler_min_time: 5 proxy_budget_rescheduler_max_time: 10 litellm_settings: drop_params: true num_retries: 3 request_timeout: 600 cache: true cache_params: type: redis host: redis port: 6379 # OTEL v2 trace destination for the logging suite's trace-completeness # tests: the arize_phoenix preset is OTLP with a configurable endpoint # (PHOENIX_COLLECTOR_HTTP_ENDPOINT below points it at the jaeger service), # so gen-AI spans export through a preset-owned provider - the code path # where trace splits actually happen - with no cloud credentials needed. callbacks: ["arize_phoenix", "datadog"] router_settings: routing_strategy: simple-shuffle num_retries: 3 allowed_fails: 5 cooldown_time: 30 fallbacks: - gemini-2.5-flash: ["gpt-5.5", "claude-haiku-4-5"] finetune_settings: - custom_llm_provider: openai api_key: os.environ/OPENAI_API_KEY files_settings: - custom_llm_provider: openai api_key: os.environ/OPENAI_API_KEY - custom_llm_provider: azure api_base: os.environ/AZURE_API_BASE api_key: os.environ/AZURE_API_KEY api_version: "2024-05-01-preview" model_list: - model_name: gpt-5.5 litellm_params: model: openai/gpt-5.5 api_key: os.environ/OPENAI_API_KEY - model_name: claude-haiku-4-5 litellm_params: model: anthropic/claude-haiku-4-5 api_key: os.environ/ANTHROPIC_API_KEY - model_name: gemini-2.5-flash litellm_params: model: gemini/gemini-2.5-flash api_key: os.environ/GEMINI_API_KEY - model_name: openai-text-embedding-3-small litellm_params: model: openai/text-embedding-3-small api_key: os.environ/OPENAI_API_KEY # v2 auto-router with the LLM complexity classifier. SIMPLE stays on the # openai backend; every higher tier routes to the anthropic backend, so the # served deployment (read back from the spend log's model) reveals whether # the LLM classifier actually ran or silently fell back to heuristic scoring. - model_name: complexity-smart-router litellm_params: model: auto_router/complexity_router complexity_router_config: classifier_type: llm classifier_llm_config: model: gpt-5.5 tiers: SIMPLE: gpt-5.5 MEDIUM: claude-haiku-4-5 COMPLEX: claude-haiku-4-5 REASONING: claude-haiku-4-5 services: litellm: image: ghcr.io/berriai/litellm:main-latest depends_on: db: condition: service_healthy redis: condition: service_healthy jaeger: condition: service_healthy dd-sink: condition: service_healthy env_file: .env environment: LITELLM_MASTER_KEY: sk-1234 STORE_MODEL_IN_DB: "True" DD_API_KEY: local-sink-noauth DD_SITE: datadoghq.com DD_BASE_URL: http://dd-sink:8080 LITELLM_OTEL_V2: "true" PHOENIX_COLLECTOR_HTTP_ENDPOINT: http://jaeger:4318/v1/traces PHOENIX_API_KEY: local-jaeger-noauth DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm UI_USERNAME: admin UI_PASSWORD: sk-1234 AWS_S3_BUCKET_NAME: ${AWS_S3_BUCKET_NAME:-${AWS_BATCH_S3_BUCKET:-}} AWS_BATCH_S3_BUCKET: ${AWS_BATCH_S3_BUCKET:-${AWS_S3_BUCKET_NAME:-}} AWS_BATCH_ROLE_ARN: ${AWS_BATCH_ROLE_ARN:-} AWS_ACCESS_KEY_ID: ${AWS_ACCESS_KEY_ID:-} AWS_SECRET_ACCESS_KEY: ${AWS_SECRET_ACCESS_KEY:-} AWS_REGION: ${AWS_REGION:-us-east-1} GCS_BUCKET_NAME: ${GCS_BUCKET_NAME:-} VERTEXAI_PROJECT: ${VERTEXAI_PROJECT:-} VERTEXAI_CREDENTIALS: ${VERTEXAI_CREDENTIALS:-} GOOGLE_APPLICATION_CREDENTIALS: ${GOOGLE_APPLICATION_CREDENTIALS:-} MISTRAL_API_KEY: ${MISTRAL_API_KEY:-} AZURE_API_BASE: ${AZURE_API_BASE:-} AZURE_API_KEY: ${AZURE_API_KEY:-} AZURE_AI_API_BASE: ${AZURE_AI_API_BASE:-} AZURE_AI_API_KEY: ${AZURE_AI_API_KEY:-} ports: - "4000:4000" configs: - source: litellm_config target: /app/config.yaml command: ["--config", "/app/config.yaml", "--port", "4000"] # throwaway db db: image: postgres:16 environment: POSTGRES_USER: litellm POSTGRES_PASSWORD: litellm POSTGRES_DB: litellm healthcheck: test: ["CMD-SHELL", "pg_isready -U litellm"] interval: 3s timeout: 3s retries: 20 redis: image: redis:7 healthcheck: test: ["CMD", "redis-cli", "ping"] interval: 3s timeout: 3s retries: 20 # throwaway OTEL trace destination (OTLP ingest on 4318 inside the network, # query API on host 16686 for test read-back; see E2E_OTEL_QUERY_URL) jaeger: image: jaegertracing/all-in-one:1.62.0 ports: - "16686:16686" healthcheck: test: ["CMD", "wget", "-qO-", "http://localhost:14269/"] interval: 3s timeout: 3s retries: 20 # throwaway DataDog logs-intake sink (records POSTs, replays on GET /requests; # see E2E_DD_SINK_URL) dd-sink: image: python:3.12-alpine command: ["python", "/sink.py"] configs: - source: dd_sink_script target: /sink.py ports: - "9915:8080" healthcheck: test: ["CMD", "wget", "-qO-", "http://127.0.0.1:8080/health"] interval: 3s timeout: 3s retries: 20