mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
The gateway and backend probes omitted timeoutSeconds, so kubelet applied its 1s default. Both containers run a single uvicorn worker (the gateway defaults NUM_WORKERS to 1; the backend passes no --workers at all), so each pod is one asyncio event loop and its per-request latency under closed-loop saturation rises by queueing (~57-62ms serial vs ~6s at 100 concurrent users against one replica). Both /health/readiness and /health/liveliness then time out on the stage cluster while the pod is serving traffic correctly, which exposes the deployment to losing a healthy pod from its load balancer during a burst and to restarting a merely busy one. Readiness now gets timeoutSeconds 10, equal to periodSeconds and above the measured saturated latency, and keeps failureThreshold 3. kubelet drives each probe from a time.Ticker of periodSeconds rather than sleeping between attempts, and coalesces ticks that arrive mid-probe, so the interval between probe starts is max(periodSeconds, probeDuration) and not their sum. Keeping timeoutSeconds <= periodSeconds is what holds that interval at 10s, so three consecutive failures still evict a genuinely wedged pod in ~30s. Liveness gets the same timeout plus failureThreshold 6: /health/liveliness is an in-memory flag check, so a timeout there only ever means event-loop starvation, which a restart makes worse, and it now needs ~90s of sustained unresponsiveness to fire. The ui container keeps the default. It is nginx serving a Next.js static export, so / is a file off disk with no application runtime that could queue behind saturated work, and nothing measured suggests it needs more than 1s.
407 lines
14 KiB
YAML
407 lines
14 KiB
YAML
# LiteLLM helm chart values
|
|
|
|
nameOverride: ""
|
|
fullnameOverride: ""
|
|
|
|
imagePullSecrets: []
|
|
|
|
# Optional Ingress wiring the three component Services behind a single L7
|
|
# entrypoint. Required when serving the static UI bundle over the network.
|
|
ingress:
|
|
enabled: false
|
|
className: ""
|
|
annotations: {}
|
|
host: "" # optional; if set, becomes the rule's host
|
|
tls: []
|
|
|
|
# Per-component ServiceAccounts for gateway, backend, and ui.
|
|
#
|
|
# Each section mirrors the old shared serviceAccount shape. Set `create:
|
|
# true` to have the chart provision the SA (useful for EKS Pod Identity /
|
|
# GKE Workload Identity annotations). Set `name` to bind an existing SA.
|
|
# When both are unset the component pod runs with the namespace `default` SA.
|
|
#
|
|
# The UI SA deliberately defaults to `automount: false` — the static nginx
|
|
# container does not need the K8s API and should not carry a projected
|
|
# ServiceAccount token that a compromised container could use to call the
|
|
# cloud-provider metadata service or the K8s API.
|
|
serviceAccounts:
|
|
gateway:
|
|
create: false
|
|
automount: true
|
|
annotations: {}
|
|
name: ""
|
|
backend:
|
|
create: false
|
|
automount: true
|
|
annotations: {}
|
|
name: ""
|
|
ui:
|
|
create: false
|
|
automount: false
|
|
annotations: {}
|
|
name: ""
|
|
|
|
# Pre-install / pre-upgrade Helm hook that runs `prisma migrate deploy`
|
|
# against the writer database, creating the LiteLLM schema (tables that
|
|
# gateway + backend assume exist at startup: LiteLLM_Config,
|
|
# LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...). Disable if your
|
|
# pipeline runs migrations out-of-band.
|
|
#
|
|
# Uses a dedicated `litellm-migrations` image (prisma CLI + the migration
|
|
# files from `litellm-proxy-extras`) instead of the backend image, so the
|
|
# Job doesn't drag in the rest of the proxy and doesn't run `prisma
|
|
# generate` — the migration engine doesn't need the generated client.
|
|
migrationJob:
|
|
enabled: true
|
|
backoffLimit: 4
|
|
ttlSecondsAfterFinished: 120
|
|
resources: {}
|
|
# ServiceAccount for the Job pod only.
|
|
#
|
|
# The Job is a pre-install / pre-upgrade hook, so it runs before the chart's
|
|
# ordinary resources exist. With `serviceAccounts.backend.create: true` the
|
|
# backend ServiceAccount is one of those ordinary resources, so a Job that
|
|
# borrowed its name would reference an account that does not exist yet and
|
|
# the first install would fail with a forbidden pod creation. The name set
|
|
# here always wins; when it is empty the Job falls back to `default` if the
|
|
# chart creates the backend ServiceAccount, and to the backend
|
|
# ServiceAccount name otherwise (that name is either an existing account you
|
|
# supplied or `default`).
|
|
#
|
|
# Point this at a pre-existing ServiceAccount when the Job needs credentials
|
|
# of its own, e.g. the IRSA / Workload Identity annotations that
|
|
# `database.writer.useIAMAuth` relies on. That is also the upgrade path to
|
|
# watch: a release already running with `serviceAccounts.backend.create:
|
|
# true` used to hand the Job the created backend account on every upgrade,
|
|
# and now hands it `default` unless you name an account here.
|
|
serviceAccountName: ""
|
|
# The Job runs `prisma migrate deploy` against Postgres and never calls the
|
|
# K8s API, so it defaults to no projected ServiceAccount token, the same
|
|
# reasoning the ui SA above uses. Flip to true if your Job genuinely needs
|
|
# one; IAM database auth does not, since EKS Pod Identity injects its own
|
|
# projected token volume and GKE Workload Identity goes through the
|
|
# metadata server, neither of which is the default token mount.
|
|
automountServiceAccountToken: false
|
|
# Standard k8s pod-level and container-level securityContext for the Job
|
|
# pod. Same shape as gateway.podSecurityContext / gateway.securityContext.
|
|
podSecurityContext: {}
|
|
securityContext: {}
|
|
# Extra pod labels on the Job pod, merged into the chart's common labels.
|
|
podLabels: {}
|
|
# Additional volumes on the Job pod and volumeMounts on its container, e.g.
|
|
# the writable scratch space a read-only root filesystem needs.
|
|
volumes: []
|
|
volumeMounts: []
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-migrations
|
|
tag: "" # defaults to .Chart.AppVersion
|
|
pullPolicy: IfNotPresent
|
|
# Extra env appended to the migration container. The migration entrypoint
|
|
# uses the v2 resolver by default (no diff-and-force recovery — avoids the
|
|
# schema thrashing seen during rolling deploys). To opt back into the v1
|
|
# resolver, append `- name: USE_V2_MIGRATION_RESOLVER` / `value: "false"`.
|
|
extraEnv: []
|
|
|
|
# Required: a master key used by gateway + backend to mint/verify proxy tokens.
|
|
# Must reference an existing Secret.
|
|
masterKey:
|
|
secretName: litellm-master-key-secret # name of a Secret containing the master key
|
|
secretKey: master-key
|
|
|
|
# Optional: enterprise billable-request metering. When enabled, the gateway and
|
|
# backend count successful requests to inference, MCP, and A2A endpoints and push
|
|
# them to LiteLLM's collector over mutual TLS. Both components serve billable
|
|
# routes: the backend keeps the named-server MCP transport. Requires an
|
|
# enterprise license. The client certificate identifies the deployment, so it is
|
|
# mounted read-only from an existing Secret and never passed through the env.
|
|
billingMetrics:
|
|
enabled: false
|
|
endpoint: https://telemetry.litellm.ai # collector to push the counter to
|
|
# An existing Secret holding the client certificate under tls.crt and its key
|
|
# under tls.key, usually created from the onboarding artifact. The default is
|
|
# the conventional name, so the common path is to create that Secret and set
|
|
# enabled: true. Override only if yours is named differently.
|
|
secretName: litellm-billing-metrics-mtls
|
|
# Only for private or test collectors whose server certificate is not on the
|
|
# public web PKI. The production collector needs no CA override.
|
|
caSecretName: "" # existing Secret holding ca.crt
|
|
exportIntervalMs: "" # push cadence; the proxy defaults to 60000
|
|
|
|
# External Postgres connection.
|
|
database:
|
|
writer:
|
|
host: ""
|
|
port: 5432
|
|
dbname: ""
|
|
schema: ""
|
|
useIAMAuth: false
|
|
passwordSecret:
|
|
name: litellm-writer-secret
|
|
usernameKey: username
|
|
passwordKey: password
|
|
|
|
# Optional read-replica routing. When `reader.host` is set, the proxy routes
|
|
# reads (find_*, count, group_by, query_raw/_first) to this endpoint while
|
|
# writes stay on the writer. Leave `reader.host` empty to disable.
|
|
reader:
|
|
host: ""
|
|
port: 5432
|
|
dbname: ""
|
|
schema: ""
|
|
useIAMAuth: false
|
|
passwordSecret:
|
|
name: litellm-reader-secret
|
|
usernameKey: username
|
|
passwordKey: password
|
|
|
|
# Optional Redis. Leave host empty to disable.
|
|
#
|
|
# This is the proxy's coordination store: cross-pod tpm/rpm rate limits, spend
|
|
# tracking, and the pod lock manager. The chart emits REDIS_HOST / REDIS_PORT /
|
|
# REDIS_PASSWORD, which the proxy picks up through its coordination Redis env
|
|
# fallback. Response caching is separate and off unless you enable it in
|
|
# `proxy_config.litellm_settings.cache`.
|
|
#
|
|
# For full control, define `general_settings.coordination_redis` in
|
|
# `proxy_config` (host/port/password/username/url/ssl/startup_nodes/
|
|
# sentinel_nodes/sentinel_password/service_name, each accepting os.environ/VAR
|
|
# refs). An explicit block overrides these env vars.
|
|
#
|
|
# Set `cluster: true` for Redis Cluster mode (e.g. AWS ElastiCache Cluster,
|
|
# self-hosted Redis Cluster). The chart emits REDIS_CLUSTER_NODES from
|
|
# `host` / `port` as the single seed; the cluster client discovers the
|
|
# remaining nodes from CLUSTER SLOTS at startup.
|
|
redis:
|
|
cluster: false
|
|
host: ""
|
|
port: 6379
|
|
passwordSecret:
|
|
name: "" # Leave empty for auth-less Redis
|
|
passwordKey: password
|
|
|
|
# ---------- gateway (LLM data plane) ----------
|
|
gateway:
|
|
enabled: true
|
|
logLevel: INFO
|
|
# Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS,
|
|
# consumed by the gateway image entrypoint. Default is 1.
|
|
numWorkers: 1
|
|
extraEnv: [] # Add extra environment variables to the gateway
|
|
envConfigMaps: [] # Add extra environment variables to the gateway from config maps
|
|
envSecrets: [] # Add extra environment variables to the gateway from secrets
|
|
# Additional volumes on the gateway Deployment (e.g. a ConfigMap holding
|
|
# custom callback / SSO handler code, mounted next to the proxy config).
|
|
volumes: []
|
|
# Additional volumeMounts on the gateway container.
|
|
volumeMounts: []
|
|
config:
|
|
create: true
|
|
proxy_config: {}
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-gateway
|
|
tag: "" # defaults to .Chart.AppVersion
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 4000
|
|
resources:
|
|
requests:
|
|
cpu: "1"
|
|
memory: 4Gi
|
|
limits:
|
|
cpu: "2"
|
|
memory: 4Gi
|
|
livenessProbe:
|
|
httpGet: { path: /health/liveliness, port: http }
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 15
|
|
timeoutSeconds: 10
|
|
failureThreshold: 6
|
|
readinessProbe:
|
|
httpGet: { path: /health/readiness, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 10
|
|
timeoutSeconds: 10
|
|
hpa:
|
|
enabled: true
|
|
minReplicas: 1
|
|
maxReplicas: 10
|
|
targetCPUUtilizationPercentage: 70
|
|
targetMemoryUtilizationPercentage: 80
|
|
# PodDisruptionBudget for the gateway pods. Set exactly one of
|
|
# `minAvailable` / `maxUnavailable` (minAvailable wins if both are set;
|
|
# enabling without either falls back to `maxUnavailable: 1`). Disabled by
|
|
# default: with the default hpa.minReplicas of 1, a `minAvailable: 1` PDB
|
|
# would block node drains entirely.
|
|
pdb:
|
|
enabled: false
|
|
minAvailable: ""
|
|
maxUnavailable: ""
|
|
podAnnotations: {}
|
|
# Extra pod labels, merged into the chart's selector labels. Do not
|
|
# re-declare `app.kubernetes.io/name` / `instance` / `component` here: they
|
|
# form the Deployment's immutable selector.
|
|
podLabels: {}
|
|
# Pod-level securityContext, applied to every container in the pod
|
|
# (runAsNonRoot, runAsUser, fsGroup, seccompProfile, ...). Empty by default
|
|
# so the cluster's own defaults keep applying to existing installs; clusters
|
|
# enforcing a restricted Pod Security Standard usually want at least
|
|
# `runAsNonRoot: true` and `seccompProfile.type: RuntimeDefault`.
|
|
podSecurityContext: {}
|
|
# Container-level securityContext for the gateway container. Empty by
|
|
# default for the same reason. Example:
|
|
# allowPrivilegeEscalation: false
|
|
# readOnlyRootFilesystem: true
|
|
# capabilities:
|
|
# drop:
|
|
# - ALL
|
|
# `readOnlyRootFilesystem: true` needs writable scratch space; supply it
|
|
# through `volumes` / `volumeMounts` above rather than expecting the chart
|
|
# to guess the paths your workload writes to.
|
|
securityContext: {}
|
|
# Extra sidecar containers appended to the gateway pod, e.g. an auth or
|
|
# egress proxy. Rendered through `tpl`, so entries may reference chart
|
|
# values and release metadata.
|
|
extraContainers: []
|
|
# Container lifecycle hooks (postStart / preStop) for the gateway container.
|
|
lifecycle: {}
|
|
# Grace period the kubelet allows between SIGTERM and SIGKILL. Leave empty
|
|
# to inherit the Kubernetes default of 30s. Set it a few seconds above the
|
|
# proxy's GRACEFUL_SHUTDOWN_TIMEOUT when you use a draining preStop hook.
|
|
terminationGracePeriodSeconds: ""
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|
|
# Standard k8s topologySpreadConstraints for the gateway pods, e.g. to
|
|
# spread replicas across zones:
|
|
# - maxSkew: 1
|
|
# topologyKey: topology.kubernetes.io/zone
|
|
# whenUnsatisfiable: ScheduleAnyway
|
|
# labelSelector:
|
|
# matchLabels:
|
|
# app.kubernetes.io/component: gateway
|
|
topologySpreadConstraints: []
|
|
|
|
# ---------- backend (UI / management API) ----------
|
|
backend:
|
|
enabled: true
|
|
logLevel: INFO
|
|
extraEnv: []
|
|
envConfigMaps: []
|
|
envSecrets: []
|
|
# Additional volumes on the backend Deployment.
|
|
volumes: []
|
|
# Additional volumeMounts on the backend container.
|
|
volumeMounts: []
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-backend
|
|
tag: ""
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 4001
|
|
resources:
|
|
requests:
|
|
cpu: "1"
|
|
memory: 4Gi
|
|
limits:
|
|
cpu: "2"
|
|
memory: 4Gi
|
|
livenessProbe:
|
|
httpGet: { path: /health/liveliness, port: http }
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 15
|
|
timeoutSeconds: 10
|
|
failureThreshold: 6
|
|
readinessProbe:
|
|
httpGet: { path: /health/readiness, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 10
|
|
timeoutSeconds: 10
|
|
hpa:
|
|
enabled: true
|
|
minReplicas: 1
|
|
maxReplicas: 4
|
|
targetCPUUtilizationPercentage: 70
|
|
# Same shape as gateway.pdb.
|
|
pdb:
|
|
enabled: false
|
|
minAvailable: ""
|
|
maxUnavailable: ""
|
|
podAnnotations: {}
|
|
# Same shape as the gateway blocks of the same name.
|
|
podLabels: {}
|
|
podSecurityContext: {}
|
|
securityContext: {}
|
|
extraContainers: []
|
|
lifecycle: {}
|
|
terminationGracePeriodSeconds: ""
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|
|
# Same shape as gateway.topologySpreadConstraints.
|
|
topologySpreadConstraints: []
|
|
|
|
# ---------- ui (Next.js static dashboard) ----------
|
|
ui:
|
|
enabled: true
|
|
logLevel: INFO
|
|
extraEnv: []
|
|
envConfigMaps: []
|
|
envSecrets: []
|
|
# Additional volumes on the ui Deployment.
|
|
volumes: []
|
|
# Additional volumeMounts on the ui container.
|
|
volumeMounts: []
|
|
image:
|
|
repository: ghcr.io/berriai/litellm-ui
|
|
tag: ""
|
|
pullPolicy: IfNotPresent
|
|
service:
|
|
type: ClusterIP
|
|
port: 3000
|
|
# The dashboard expects to know where to reach the backend API. Set this to
|
|
# the externally-routable URL (typically the ingress host + /api or similar).
|
|
backendUrl: ""
|
|
resources:
|
|
requests:
|
|
cpu: 500m
|
|
memory: 500Mi
|
|
limits:
|
|
cpu: "1"
|
|
memory: 1Gi
|
|
livenessProbe:
|
|
httpGet: { path: /, port: http }
|
|
initialDelaySeconds: 5
|
|
periodSeconds: 20
|
|
readinessProbe:
|
|
httpGet: { path: /, port: http }
|
|
initialDelaySeconds: 2
|
|
periodSeconds: 10
|
|
hpa:
|
|
enabled: false
|
|
minReplicas: 1
|
|
maxReplicas: 3
|
|
targetCPUUtilizationPercentage: 80
|
|
# Same shape as gateway.pdb.
|
|
pdb:
|
|
enabled: false
|
|
minAvailable: ""
|
|
maxUnavailable: ""
|
|
podAnnotations: {}
|
|
# Same shape as the gateway blocks of the same name. The nginx runtime
|
|
# writes its pid, cache, and proxy temp files under the image's root
|
|
# filesystem, so `securityContext.readOnlyRootFilesystem: true` here needs
|
|
# emptyDir volumes mounted over those paths.
|
|
podLabels: {}
|
|
podSecurityContext: {}
|
|
securityContext: {}
|
|
extraContainers: []
|
|
lifecycle: {}
|
|
terminationGracePeriodSeconds: ""
|
|
nodeSelector: {}
|
|
tolerations: []
|
|
affinity: {}
|
|
# Same shape as gateway.topologySpreadConstraints.
|
|
topologySpreadConstraints: []
|