mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-13 23:11:40 +00:00
Split the monolithic LiteLLM proxy into independently scalable Kubernetes components to allow separate horizontal scaling of the LLM data plane and management API surfaces - Add DatabaseURLSettings pydantic-settings model that assembles DATABASE_URL (and optional DATABASE_URL_READ_REPLICA) from discrete DATABASE_* env vars before Prisma initializes, supporting both IAM token auth (minting short-lived RDS tokens) and password auth; replaces the CLI-only path that componentized entrypoints bypass - Add gateway component (port 4000) that trims the proxy route table to the LLM data-plane surface (chat, embeddings, completions, audio, realtime, provider passthroughs, health/metrics) via an allowlist applied inside the lifespan context so plugin-registered routes are captured - Add backend component (port 4001) that exposes the management/admin surface (keys, users, teams, orgs, spend analytics, model management, SSO, audit logs) with a complementary allowlist - Add ui component — Next.js static export served by nginx (port 3000) with RSC payload routing, asset prefix aliasing, and SPA fallback for dashboard routes - Add migrations component with dedicated Dockerfile that runs prisma migrate deploy via a Helm pre-install/pre-upgrade Job, eliminating per-pod schema contention on the Prisma advisory lock - Add Helm chart (helm/litellm) with separate Deployments, Services, HPAs, and ConfigMap for each component; shared _helpers.tpl emits DATABASE_*, IAM_TOKEN_DB_AUTH, REDIS_*, and DISABLE_SCHEMA_UPDATE env vars from chart values; ingress template routes traffic to the correct component by path prefix - Add comprehensive tests for DatabaseURLSettings covering IAM auth, password auth, read replica fallbacks, operator-pinned URL preservation, and percent-encoding; add coverage test asserting gateway + backend allowlist union equals the full proxy route set - Add pydantic-settings>=2.14.1 as a proxy extra dependency and update liccheck allowlist Co-authored-by: Yassin Kortam <yassinkortam@g.ucla.edu>
225 lines
6.2 KiB
YAML
225 lines
6.2 KiB
YAML
# LiteLLM helm chart values
|
|
|
|
nameOverride: ""
|
|
fullnameOverride: ""
|
|
|
|
imagePullSecrets: []
|
|
|
|
# Optional Ingress wiring the three component Services behind a single L7
|
|
# entrypoint. Required when serving the static UI bundle over the network.
|
|
ingress:
|
|
enabled: false
|
|
className: ""
|
|
annotations: {}
|
|
host: "" # optional; if set, becomes the rule's host
|
|
tls: []
|
|
|
|
# Shared ServiceAccount used by all three component Deployments. Set
|
|
# `create: true` to have the chart provision it (e.g. when wiring an EKS
|
|
# Pod Identity association by SA name). Set `name` to use an existing SA
|
|
# (chart-created or out-of-band). When both are empty / false, pods run
|
|
# with the namespace's `default` SA.
|
|
serviceAccount:
|
|
create: false
|
|
automount: true
|
|
annotations: {}
|
|
name: ""
|
|
|
|
# Pre-install / pre-upgrade Helm hook that runs `prisma migrate deploy`
|
|
# against the writer database, creating the LiteLLM schema (tables that
|
|
# gateway + backend assume exist at startup: LiteLLM_Config,
|
|
# LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...). Disable if your
|
|
# pipeline runs migrations out-of-band.
|
|
#
|
|
# Uses a dedicated `litellm-migrations` image (prisma CLI + the migration
|
|
# files from `litellm-proxy-extras`) instead of the backend image, so the
|
|
# Job doesn't drag in the rest of the proxy and doesn't run `prisma
|
|
# generate` — the migration engine doesn't need the generated client.
|
|
migrationJob:
|
|
enabled: true
|
|
backoffLimit: 4
|
|
ttlSecondsAfterFinished: 120
|
|
resources: {}
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-migrations
|
|
tag: "" # defaults to .Chart.AppVersion
|
|
pullPolicy: IfNotPresent
|
|
# Extra env appended to the migration container. The migration entrypoint
|
|
# uses the v2 resolver by default (no diff-and-force recovery — avoids the
|
|
# schema thrashing seen during rolling deploys). To opt back into the v1
|
|
# resolver, append `- name: USE_V2_MIGRATION_RESOLVER` / `value: "false"`.
|
|
extraEnv: []
|
|
|
|
# Required: a master key used by gateway + backend to mint/verify proxy tokens.
|
|
# Must reference an existing Secret.
|
|
masterKey:
|
|
secretName: litellm-master-key-secret # name of a Secret containing the master key
|
|
secretKey: master-key
|
|
|
|
# External Postgres connection.
|
|
database:
|
|
writer:
|
|
host: ""
|
|
port: 5432
|
|
dbname: ""
|
|
schema: ""
|
|
useIAMAuth: false
|
|
passwordSecret:
|
|
name: litellm-writer-secret
|
|
usernameKey: username
|
|
passwordKey: password
|
|
|
|
# Optional read-replica routing. When `reader.host` is set, the proxy routes
|
|
# reads (find_*, count, group_by, query_raw/_first) to this endpoint while
|
|
# writes stay on the writer. Leave `reader.host` empty to disable.
|
|
reader:
|
|
host: ""
|
|
port: 5432
|
|
dbname: ""
|
|
schema: ""
|
|
useIAMAuth: false
|
|
passwordSecret:
|
|
name: litellm-reader-secret
|
|
usernameKey: username
|
|
passwordKey: password
|
|
|
|
# Optional Redis (caching, rate limiting). Leave host empty to disable.
|
|
#
|
|
# Set `cluster: true` for Redis Cluster mode (e.g. AWS ElastiCache Cluster,
|
|
# self-hosted Redis Cluster). The chart emits REDIS_CLUSTER_NODES from
|
|
# `host` / `port` as the single seed; the cluster client discovers the
|
|
# remaining nodes from CLUSTER SLOTS at startup.
|
|
redis:
|
|
cluster: false
|
|
host: ""
|
|
port: 6379
|
|
passwordSecret:
|
|
name: "" # Leave empty for auth-less Redis
|
|
passwordKey: password
|
|
|
|
# ---------- gateway (LLM data plane) ----------
|
|
gateway:
|
|
enabled: true
|
|
logLevel: INFO
|
|
# Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS,
|
|
# consumed by the gateway image entrypoint. Default is 1.
|
|
numWorkers: 1
|
|
extraEnv: [] # Add extra environment variables to the gateway
|
|
envConfigMaps: [] # Add extra environment variables to the gateway from config maps
|
|
envSecrets: [] # Add extra environment variables to the gateway from secrets
|
|
config:
|
|
create: true
|
|
proxy_config: {}
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-gateway
|
|
tag: "" # defaults to .Chart.AppVersion
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 4000
|
|
resources:
|
|
requests:
|
|
cpu: "1"
|
|
memory: 4Gi
|
|
limits:
|
|
cpu: "2"
|
|
memory: 4Gi
|
|
livenessProbe:
|
|
httpGet: { path: /health/liveliness, port: http }
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 15
|
|
readinessProbe:
|
|
httpGet: { path: /health/readiness, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 10
|
|
hpa:
|
|
enabled: true
|
|
minReplicas: 1
|
|
maxReplicas: 10
|
|
targetCPUUtilizationPercentage: 70
|
|
targetMemoryUtilizationPercentage: 80
|
|
podAnnotations: {}
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|
|
|
|
# ---------- backend (UI / management API) ----------
|
|
backend:
|
|
enabled: true
|
|
logLevel: INFO
|
|
extraEnv: []
|
|
envConfigMaps: []
|
|
envSecrets: []
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-backend
|
|
tag: ""
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 4001
|
|
resources:
|
|
requests:
|
|
cpu: "1"
|
|
memory: 4Gi
|
|
limits:
|
|
cpu: "2"
|
|
memory: 4Gi
|
|
livenessProbe:
|
|
httpGet: { path: /health/liveliness, port: http }
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 15
|
|
readinessProbe:
|
|
httpGet: { path: /health/readiness, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 10
|
|
hpa:
|
|
enabled: true
|
|
minReplicas: 1
|
|
maxReplicas: 4
|
|
targetCPUUtilizationPercentage: 70
|
|
podAnnotations: {}
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|
|
|
|
# ---------- ui (Next.js static dashboard) ----------
|
|
ui:
|
|
enabled: true
|
|
logLevel: INFO
|
|
extraEnv: []
|
|
envConfigMaps: []
|
|
envSecrets: []
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-ui
|
|
tag: ""
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 3000
|
|
# The dashboard expects to know where to reach the backend API. Set this to
|
|
# the externally-routable URL (typically the ingress host + /api or similar).
|
|
backendUrl: ""
|
|
resources:
|
|
requests:
|
|
cpu: 500m
|
|
memory: 500Mi
|
|
limits:
|
|
cpu: "1"
|
|
memory: 1Gi
|
|
livenessProbe:
|
|
httpGet: { path: /, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 20
|
|
readinessProbe:
|
|
httpGet: { path: /, port: http }
|
|
initialDelaySeconds: 2
|
|
periodSeconds: 10
|
|
hpa:
|
|
enabled: false
|
|
minReplicas: 1
|
|
maxReplicas: 3
|
|
targetCPUUtilizationPercentage: 80
|
|
podAnnotations: {}
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|