mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Updated application-kube-prometheus-stack.yaml to pin the targetRevision to kube-prometheus-stack-83.7.0 for stability. Disabled the migration job in litellm-values.yaml to streamline deployment, allowing the proxy image to handle DB setup on startup.
124 lines
3.2 KiB
YAML
124 lines
3.2 KiB
YAML
# Helm values for EKS performance stack: LiteLLM + Bitnami Postgres + Bitnami Redis (chart subcharts).
|
|
# Argo CD: deploy/performance/argocd/application-litellm.yaml
|
|
#
|
|
# Override postgresql.auth and redis.auth passwords for non-dev clusters.
|
|
#
|
|
# Chart default image tag is main-<Chart.appVersion> (e.g. main-v1.80.12). That pattern is not always
|
|
# published for ghcr.io/berriai/litellm-database; use main-latest or a tag from:
|
|
# https://github.com/BerriAI/litellm/pkgs/container/litellm-database
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-database
|
|
tag: main-latest
|
|
|
|
replicaCount: 2
|
|
|
|
args:
|
|
- --config
|
|
- /etc/litellm/config.yaml
|
|
- --num_workers
|
|
- "2"
|
|
- --max_requests_before_restart
|
|
- "20000"
|
|
|
|
service:
|
|
type: LoadBalancer
|
|
port: 4000
|
|
|
|
nodeSelector:
|
|
workload: litellm
|
|
|
|
securityContext:
|
|
capabilities:
|
|
add:
|
|
- SYS_PTRACE
|
|
|
|
resources:
|
|
requests:
|
|
cpu: "500m"
|
|
memory: "1Gi"
|
|
limits:
|
|
cpu: "4"
|
|
memory: "4Gi"
|
|
|
|
readinessProbe:
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 10
|
|
timeoutSeconds: 5
|
|
failureThreshold: 6
|
|
|
|
livenessProbe:
|
|
initialDelaySeconds: 45
|
|
periodSeconds: 30
|
|
timeoutSeconds: 10
|
|
failureThreshold: 3
|
|
|
|
startupProbe:
|
|
failureThreshold: 30
|
|
|
|
envVars:
|
|
STORE_MODEL_IN_DB: "True"
|
|
ENV: "production"
|
|
LITELLM_ENVIRONMENT: "production"
|
|
|
|
db:
|
|
deployStandalone: true
|
|
useExisting: false
|
|
|
|
# Bitnami subcharts: set StorageClass so PVCs provision on EKS (e.g. gp2 in-tree or gp3 via CSI).
|
|
# Without this, claims stay Pending with an empty STORAGECLASS. If you already have Pending PVCs,
|
|
# delete them after syncing so StatefulSets recreate claims (wiping local DB/Redis data).
|
|
postgresql:
|
|
# docker.io/bitnami/postgresql tags from the subchart (e.g. 16.2.0-debian-12-r6) are often removed;
|
|
# use Bitnami Legacy Catalog: https://hub.docker.com/r/bitnamilegacy/postgresql/tags
|
|
image:
|
|
registry: docker.io
|
|
repository: bitnamilegacy/postgresql
|
|
tag: 16.6.0-debian-12-r2
|
|
primary:
|
|
persistence:
|
|
storageClass: gp2
|
|
|
|
redis:
|
|
enabled: true
|
|
architecture: standalone
|
|
# Same as Postgres: chart default docker.io/bitnami/redis tags may 404; legacy catalog:
|
|
# https://hub.docker.com/r/bitnamilegacy/redis/tags
|
|
image:
|
|
registry: docker.io
|
|
repository: bitnamilegacy/redis
|
|
tag: 7.2.5-debian-12-r6
|
|
master:
|
|
persistence:
|
|
storageClass: gp2
|
|
|
|
# Skip separate Batch Job + Argo PreSync: proxy image runs DB setup on startup when
|
|
# DISABLE_SCHEMA_UPDATE is not set (see litellm/proxy/proxy_cli.py). Use the Job again if you
|
|
# prefer migrations before any proxy pod starts, or if you run many replicas and want a single migration step.
|
|
migrationJob:
|
|
enabled: false
|
|
hooks:
|
|
argocd:
|
|
enabled: false
|
|
|
|
proxy_config:
|
|
model_list:
|
|
- model_name: fake-openai-endpoint
|
|
litellm_params:
|
|
model: openai/fake-model
|
|
api_key: fake-key
|
|
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
|
timeout: 40
|
|
general_settings:
|
|
master_key: os.environ/PROXY_MASTER_KEY
|
|
health_check_details: false
|
|
database_url: os.environ/DATABASE_URL
|
|
litellm_settings:
|
|
drop_params: true
|
|
telemetry: false
|
|
callbacks: []
|
|
cache: true
|
|
cache_params:
|
|
type: redis
|
|
supported_call_types: []
|
|
|
|
extraResources: []
|