litellm/deploy/aws/locustfile.py
Julio Quinteros Pro 445c67cfec Add AWS ECS deployment template matching benchmark specifications
This commit adds a complete 1-click deployment solution for LiteLLM on AWS ECS,
configured to match the benchmark specifications from https://docs.litellm.ai/docs/benchmarks

## What's Added

### Infrastructure (1 file)
- cloudformation-ecs.yaml: AWS CloudFormation template for ECS deployment
  - 4 ECS Fargate tasks (4 vCPU, 8 GB RAM each)
  - 4 workers per task (16 total workers)
  - RDS PostgreSQL database (db.t3.medium)
  - Application Load Balancer
  - VPC with public/private subnets across 2 AZs
  - Security groups, NAT Gateway, monitoring

### Deployment Tools (3 files)
- deploy.sh: Automated deployment script with interactive prompts
- test-deployment.sh: Deployment validation and health check script
- cost-calculator.sh: Interactive cost estimation tool

### Documentation (6 files)
- 00-START-HERE.md: Quick start guide and overview
- QUICKSTART.md: 5-minute deployment guide
- README.md: Complete deployment documentation
- ARCHITECTURE.md: Detailed architecture deep-dive with diagrams
- INDEX.md: Master index of all files
- .summary.md: Internal summary document

### Testing & Configuration (2 files)
- locustfile.py: Load testing script to replicate benchmark tests
- example-config.yaml: LiteLLM configuration example

## Configuration

- 4 instances with 4 vCPU and 8 GB RAM each
- 4 workers per instance
- Expected performance:
  - Median latency: ~100 ms
  - P95 latency: ~150 ms
  - Throughput: ~1,170 RPS
  - LiteLLM overhead: ~2 ms

## Usage

```bash
cd deploy/aws
./deploy.sh
```

## Monthly Cost

~$440-460 (pay-as-you-go) or ~$270-370 (with reserved capacity)

## Features

- ✅ CloudFormation template validated with AWS
- ✅ Production-ready with high availability
- ✅ Secure by default (private subnets, security groups, encrypted secrets)
- ✅ Well-documented with comprehensive guides
- ✅ Includes validation and load testing tools
- ✅ Cost-optimized configuration

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
2026-02-16 13:03:05 -03:00

231 lines
7.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
LiteLLM Benchmark Load Testing with Locust
This script replicates the benchmark testing described in:
https://docs.litellm.ai/docs/benchmarks
Usage:
# Set environment variables
export LITELLM_HOST="http://your-load-balancer-url"
export LITELLM_MASTER_KEY="your-master-key"
# Run with benchmark parameters (1000 users, 500 spawn rate, 5 minutes)
locust -f locustfile.py --host=$LITELLM_HOST --users=1000 --spawn-rate=500 --run-time=5m --headless
# Run with web UI for interactive testing
locust -f locustfile.py --host=$LITELLM_HOST
# Run with custom parameters
locust -f locustfile.py --host=$LITELLM_HOST --users=500 --spawn-rate=100 --run-time=10m --headless
"""
import os
import time
import json
from locust import HttpUser, task, between, events
from locust.runners import MasterRunner
class LiteLLMUser(HttpUser):
"""
Simulates a user making requests to LiteLLM proxy server.
"""
# Wait time between tasks (benchmark uses continuous load)
wait_time = between(0.1, 0.5)
def on_start(self):
"""
Called when a simulated user starts.
Sets up authentication and headers.
"""
self.master_key = os.environ.get("LITELLM_MASTER_KEY")
if not self.master_key:
raise ValueError(
"LITELLM_MASTER_KEY environment variable is required. "
"Set it with: export LITELLM_MASTER_KEY='your-key'"
)
self.headers = {
"Authorization": f"Bearer {self.master_key}",
"Content-Type": "application/json",
}
@task(10)
def chat_completion(self):
"""
Main task: Send chat completion request to LiteLLM.
This is weighted at 10 to be the primary task.
"""
payload = {
"model": "fake-openai-endpoint",
"messages": [
{"role": "user", "content": "Hello, how are you?"}
],
}
with self.client.post(
"/v1/chat/completions",
headers=self.headers,
json=payload,
catch_response=True,
name="Chat Completion"
) as response:
if response.status_code == 200:
# Check for LiteLLM overhead header
overhead = response.headers.get("x-litellm-overhead-duration-ms")
if overhead:
# Record custom metric for LiteLLM overhead
events.request.fire(
request_type="OVERHEAD",
name="LiteLLM Overhead (ms)",
response_time=float(overhead),
response_length=0,
exception=None,
context={}
)
response.success()
else:
response.failure(f"Failed with status {response.status_code}: {response.text}")
@task(1)
def health_check(self):
"""
Health check task to verify service is running.
This is weighted at 1 to run occasionally.
"""
with self.client.get(
"/health/readiness",
catch_response=True,
name="Health Check"
) as response:
if response.status_code == 200:
response.success()
else:
response.failure(f"Health check failed: {response.status_code}")
@task(5)
def streaming_completion(self):
"""
Streaming chat completion request.
This is weighted at 5 to run less frequently than regular completions.
"""
payload = {
"model": "fake-openai-endpoint",
"messages": [
{"role": "user", "content": "Tell me a short story"}
],
"stream": True,
}
with self.client.post(
"/v1/chat/completions",
headers=self.headers,
json=payload,
catch_response=True,
stream=True,
name="Streaming Completion"
) as response:
if response.status_code == 200:
# Consume the stream
for chunk in response.iter_lines():
if chunk:
pass # Process chunks if needed
response.success()
else:
response.failure(f"Streaming failed: {response.status_code}")
class BenchmarkUser(HttpUser):
"""
Simplified user class for pure benchmark testing.
This mimics the exact behavior from the benchmark guide.
"""
wait_time = between(0, 0.1) # Minimal wait time for maximum load
def on_start(self):
self.master_key = os.environ.get("LITELLM_MASTER_KEY")
if not self.master_key:
raise ValueError("LITELLM_MASTER_KEY environment variable is required")
self.headers = {
"Authorization": f"Bearer {self.master_key}",
"Content-Type": "application/json",
}
@task
def benchmark_request(self):
"""
Single benchmark request matching the benchmark guide.
"""
payload = {
"model": "fake-openai-endpoint",
"messages": [{"role": "user", "content": "test"}],
}
start_time = time.time()
with self.client.post(
"/v1/chat/completions",
headers=self.headers,
json=payload,
catch_response=True,
name="Benchmark Request"
) as response:
total_time = (time.time() - start_time) * 1000 # Convert to ms
if response.status_code == 200:
# Extract LiteLLM overhead
overhead = response.headers.get("x-litellm-overhead-duration-ms", "0")
litellm_overhead = float(overhead)
# Record metrics
events.request.fire(
request_type="METRIC",
name="LiteLLM Overhead",
response_time=litellm_overhead,
response_length=0,
exception=None,
context={}
)
response.success()
else:
response.failure(f"Status: {response.status_code}")
# Custom event handlers for enhanced reporting
@events.test_start.add_listener
def on_test_start(environment, **kwargs):
"""
Print test configuration when test starts.
"""
print("\n" + "=" * 60)
print("LiteLLM Benchmark Load Test")
print("=" * 60)
print(f"Host: {environment.host}")
print(f"Users: {environment.runner.target_user_count if hasattr(environment.runner, 'target_user_count') else 'N/A'}")
print("Benchmark Configuration: 4 instances × 4 workers")
print("Expected Performance:")
print(" - Median latency: ~100 ms")
print(" - P95 latency: ~150 ms")
print(" - Throughput: ~1,170 RPS")
print(" - LiteLLM overhead: ~2 ms")
print("=" * 60 + "\n")
@events.test_stop.add_listener
def on_test_stop(environment, **kwargs):
"""
Print summary when test stops.
"""
print("\n" + "=" * 60)
print("Test Completed")
print("=" * 60)
print("Compare your results with the benchmark:")
print("https://docs.litellm.ai/docs/benchmarks")
print("=" * 60 + "\n")
# Instructions for users
if __name__ == "__main__":
print(__doc__)