From d1ca826ff640c0dcdbaa5315399eefd851f46454 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Tue, 4 Aug 2026 15:17:12 -0700 Subject: [PATCH] docs(helm): replace the classic chart's 128Mi resource example with the documented 4Gi sizing (#35830) The litellm-helm values file shipped the stock helm create boilerplate for resources: an empty default plus a commented 100m/128Mi example it invites operators to uncomment. 128Mi is roughly 32x below what the proxy needs at DB-connected steady state, and it was the only sizing figure this chart ever showed, so operators who followed it were sized for OOMKills. Point the example at the documented 1 CPU / 4Gi per worker instead, link the production sizing guidance, and note why the default stays unset. The migration job's commented block carried the same trap with a 100m/100Mi example; drop those numbers rather than substitute proxy figures that do not transfer to a job that migrates and exits. The defaults are deliberately left at {} so no existing release changes shape on upgrade; rendered output is unchanged. --- helm/litellm-helm/README.md | 2 +- helm/litellm-helm/values.yaml | 27 +++++++++++++++------------ 2 files changed, 16 insertions(+), 13 deletions(-) diff --git a/helm/litellm-helm/README.md b/helm/litellm-helm/README.md index 4e0884dd08c..4c8712ea7b9 100644 --- a/helm/litellm-helm/README.md +++ b/helm/litellm-helm/README.md @@ -39,7 +39,7 @@ If `db.useStackgresOperator` is used (not yet implemented): | `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | | `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | | `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | -| `resources.*` | CPU/memory requests and limits for the LiteLLM container. | `{}` | +| `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` | | `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` | | `ingress.labels` | Additional labels for the Ingress resource | `{}` | | `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A | diff --git a/helm/litellm-helm/values.yaml b/helm/litellm-helm/values.yaml index 7235bb0bd78..df2b55723fe 100644 --- a/helm/litellm-helm/values.yaml +++ b/helm/litellm-helm/values.yaml @@ -181,16 +181,19 @@ proxy_config: resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi + # Unset by default so the chart installs on small clusters such as Minikube, and so an + # upgrade never leaves a running pod Pending. Production deployments should set these. + # A proxy at DB-connected steady state needs about 1 CPU and 4Gi of memory per worker; + # sizing below that gets the pod OOMKilled once traffic and DB connections ramp up. + # Scale both figures with --num_workers, then uncomment the lines below and remove the + # curly braces after 'resources:'. See "Recommended Machine Specifications" in + # https://docs.litellm.ai/docs/proxy/prod. # requests: - # cpu: 100m - # memory: 128Mi + # cpu: "1" + # memory: 4Gi + # limits: + # cpu: "1" + # memory: 4Gi autoscaling: enabled: false @@ -432,9 +435,9 @@ migrationJob: annotations: {} ttlSecondsAfterFinished: 120 resources: {} - # requests: - # cpu: 100m - # memory: 100Mi + # Unset by default. This job runs the database migration and exits, so it does not + # need the steady-state headroom the proxy does; size it from your own migration + # runs rather than from the proxy figures above. extraContainers: [] extraInitContainers: []