This commit is contained in:
devin-ai-integration[bot] 2026-09-28 08:26:47 +02:00 • committed by GitHub
commit 87c31c86f0
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
19 changed files with 282 additions and 6 deletions

View file

@ -36,12 +36,14 @@ If `db.useStackgresOperator` is used (not yet implemented):
| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` |
| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` |
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
| `keepaliveTimeoutSeconds` | Uvicorn keep-alive timeout in seconds. Must exceed the load balancer idle timeout in front of the gateway. | `630` |
| `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
| `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
| `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
| `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` |
| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
| `ingress.labels` | Additional labels for the Ingress resource | `{}` |
| `ingress.idleTimeoutSeconds` | Load balancer idle timeout in seconds. Renders ALB idle-timeout or nginx proxy read/send timeout annotations unless those keys are user-supplied. Set to `0` to disable. | `600` |
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
| `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` |
| `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` |

View file

@ -57,6 +57,10 @@ spec:
imagePullPolicy: {{ .Values.image.pullPolicy }}
env:
{{- include "litellm.proxyEnv" . | nindent 12 }}
{{- if and .Values.keepaliveTimeoutSeconds (not (hasKey (default dict .Values.envVars) "KEEPALIVE_TIMEOUT")) }}
- name: KEEPALIVE_TIMEOUT
value: {{ .Values.keepaliveTimeoutSeconds | quote }}
{{- end }}
{{- include "litellm.proxyMetricsEnv" . | nindent 12 }}
{{- if .Values.collector.enabled }}
{{- include "litellm.collectorEnv" . | nindent 12 }}

View file

@ -6,6 +6,17 @@
{{- $_ := set .Values.ingress.annotations "kubernetes.io/ingress.class" .Values.ingress.className}}
{{- end }}
{{- end }}
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
{{- if and (eq .Values.ingress.className "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- else if and (eq .Values.ingress.className "nginx") .Values.ingress.idleTimeoutSeconds -}}
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- end }}
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- end }}
{{- end }}
{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion -}}
apiVersion: networking.k8s.io/v1
{{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion -}}
@ -21,9 +32,9 @@ metadata:
{{- with .Values.ingress.labels }}
{{- toYaml . | nindent 4 }}
{{- end }}
{{- with .Values.ingress.annotations }}
{{- if $annotations }}
annotations:
{{- toYaml . | nindent 4 }}
{{- toYaml $annotations | nindent 4 }}
{{- end }}
spec:
{{- if and .Values.ingress.className (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion) }}

View file

@ -16,6 +16,41 @@ tests:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm:test
- it: should set KEEPALIVE_TIMEOUT by default
template: deployment.yaml
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
value: "630"
- it: should omit KEEPALIVE_TIMEOUT when configured to zero
template: deployment.yaml
set:
keepaliveTimeoutSeconds: 0
asserts:
- notContains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
any: true
- it: should preserve an envVars KEEPALIVE_TIMEOUT override without duplication
template: deployment.yaml
set:
envVars:
KEEPALIVE_TIMEOUT: "120"
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
value: "120"
- notContains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
value: "630"
any: true
- it: should work with tolerations
template: deployment.yaml
set:

View file

@ -16,6 +16,49 @@ tests:
- isKind:
of: Ingress
- it: should add the default nginx proxy timeout annotations
set:
ingress.enabled: true
asserts:
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
value: "600"
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
value: "600"
- it: should add the default ALB idle timeout annotation
set:
ingress.enabled: true
ingress.className: alb
asserts:
- equal:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
value: idle_timeout.timeout_seconds=600
- it: should preserve user-supplied timeout annotations
set:
ingress.enabled: true
ingress.annotations:
alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true"
ingress.className: alb
asserts:
- equal:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
- it: should omit timeout annotations when idle timeout is zero
set:
ingress.enabled: true
ingress.idleTimeoutSeconds: 0
asserts:
- notExists:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
- notExists:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
- notExists:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
- it: should add custom labels
set:
ingress.enabled: true

View file

@ -117,6 +117,12 @@ startupProbe:
ingress:
enabled: false
className: "nginx"
# Load balancer idle timeout in seconds. ALB renders
# alb.ingress.kubernetes.io/load-balancer-attributes:
# idle_timeout.timeout_seconds=N. Nginx renders
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
# User-supplied annotations win, and 0 disables generated annotations.
idleTimeoutSeconds: 600
labels: {}
annotations:
{}
@ -586,6 +592,10 @@ migrationJob:
# this injection entirely when envVars already defines LITELLM_LOG.
logLevel: INFO
# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the
# load balancer idle timeout in front of the gateway.
keepaliveTimeoutSeconds: 630
# Additional environment variables to be added to the deployment as a map of key-value pairs
envVars: {}

View file

@ -64,6 +64,10 @@ spec:
- name: NUM_WORKERS
value: {{ .Values.gateway.numWorkers | quote }}
{{- end }}
{{- if .Values.gateway.keepaliveTimeoutSeconds }}
- name: KEEPALIVE_TIMEOUT
value: {{ .Values.gateway.keepaliveTimeoutSeconds | quote }}
{{- end }}
{{- if .Values.database.connectionPool.enabled }}
{{- include "litellm.connectionPoolEnv" $ | nindent 12 }}
{{- end }}

View file

@ -6,6 +6,17 @@
{{- $backendPort := .Values.backend.service.port -}}
{{- $uiPort := .Values.ui.service.port -}}
{{- $controller := .Values.ingress.controller | default "alb" -}}
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
{{- if and (eq $controller "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- else if and (eq $controller "nginx") .Values.ingress.idleTimeoutSeconds -}}
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- end }}
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
{{- end }}
{{- end }}
{{- if not (has $controller (list "alb" "nginx")) }}
{{- fail (printf "ingress.controller: unknown controller %q, expected one of alb, nginx" $controller) }}
{{- end }}
@ -96,9 +107,9 @@ metadata:
name: {{ include "litellm.fullname" . }}
labels:
{{- include "litellm.commonLabels" . | nindent 4 }}
{{- with .Values.ingress.annotations }}
{{- if $annotations }}
annotations:
{{- toYaml . | nindent 4 }}
{{- toYaml $annotations | nindent 4 }}
{{- end }}
spec:
{{- with .Values.ingress.className }}

View file

@ -0,0 +1,33 @@
suite: test gateway environment defaults
templates:
- gateway/deployment.yaml
- gateway/configmap.yaml
values:
- ./values/required.yaml
tests:
- it: sets KEEPALIVE_TIMEOUT only on the gateway container
template: gateway/deployment.yaml
set:
gateway.collector.enabled: true
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
value: "630"
- notContains:
path: spec.template.spec.containers[1].env
content:
name: KEEPALIVE_TIMEOUT
any: true
- it: omits KEEPALIVE_TIMEOUT when configured to zero
template: gateway/deployment.yaml
set:
gateway.keepaliveTimeoutSeconds: 0
asserts:
- notContains:
path: spec.template.spec.containers[0].env
content:
name: KEEPALIVE_TIMEOUT
any: true

View file

@ -4,6 +4,65 @@ templates:
values:
- ./values/required.yaml
tests:
- it: adds the default ALB idle timeout annotation
set:
ingress.enabled: true
asserts:
- equal:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
value: idle_timeout.timeout_seconds=600
- it: preserves a user-supplied ALB load balancer attributes annotation
set:
ingress.enabled: true
ingress.annotations:
alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true"
asserts:
- equal:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
- it: adds nginx proxy timeout annotations for ingress-nginx
set:
ingress.enabled: true
ingress.controller: nginx
asserts:
- notExists:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
value: "600"
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
value: "600"
- it: preserves user-supplied nginx proxy timeout annotations
set:
ingress.enabled: true
ingress.controller: nginx
ingress.annotations:
nginx.ingress.kubernetes.io/proxy-read-timeout: "120"
nginx.ingress.kubernetes.io/proxy-send-timeout: "180"
asserts:
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
value: "120"
- equal:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
value: "180"
- it: does not add load balancer timeout annotations when disabled
set:
ingress.enabled: true
ingress.idleTimeoutSeconds: 0
asserts:
- notExists:
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
- notExists:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
- notExists:
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
- it: keeps the AWS Load Balancer Controller path types by default
set:
ingress.enabled: true

View file

@ -23,6 +23,12 @@ ingress:
# wildcard pathType, so that rule could never match there.
controller: alb
annotations: {}
# Load balancer idle timeout in seconds. ALB renders
# alb.ingress.kubernetes.io/load-balancer-attributes:
# idle_timeout.timeout_seconds=N. Nginx renders
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
# Streams silent longer than this (slow first token) are cut with a 504.
idleTimeoutSeconds: 600
host: "" # optional; if set, becomes the rule's host
tls: []
# Extra HTTP paths appended to the ingress rule. Additive: every built-in
@ -282,6 +288,10 @@ gateway:
# Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS,
# consumed by the gateway image entrypoint. Default is 1.
numWorkers: 1
# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the
# load balancer idle timeout in front of the gateway, otherwise the LB can
# hand a request to a connection uvicorn is closing.
keepaliveTimeoutSeconds: 630
extraEnv: [] # Add extra environment variables to the gateway
envConfigMaps: [] # Add extra environment variables to the gateway from config maps
envSecrets: [] # Add extra environment variables to the gateway from secrets

View file

@ -335,6 +335,17 @@ gateway_tokens_metric = {
}
```
### Load balancer and gateway timeouts
The ALB idle timeout defaults to 600 seconds through `alb_idle_timeout_seconds`.
The gateway receives `KEEPALIVE_TIMEOUT` set to 30 seconds above that value, so
uvicorn keeps connections open longer than the load balancer. Override
`gateway_extra_env.KEEPALIVE_TIMEOUT` when a different gateway timeout is needed.
```hcl
alb_idle_timeout_seconds = 600
```
Worked example for the request policy: 1,000 rps across 10 tasks is 100 rps
per task (the ALB reports it as 6,000 per minute per target) against a target
of 90 (5,400), so target tracking sizes the service to

View file

@ -5,7 +5,7 @@ resource "aws_lb" "this" {
security_groups = [aws_security_group.alb.id]
subnets = local.public_subnet_ids
idle_timeout = 120
idle_timeout = var.alb_idle_timeout_seconds
lifecycle {
precondition {

View file

@ -176,6 +176,9 @@ locals {
backend_extra_env_list = [
for k, v in var.backend_extra_env : { name = k, value = v }
]
gateway_timeout_env = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [
{ name = "KEEPALIVE_TIMEOUT", value = tostring(var.alb_idle_timeout_seconds + 30) },
]
# Storing models in the DB needs a DB. Without one the backend reads its
# model list from proxy_config only.
@ -292,6 +295,7 @@ locals {
local.shared_env,
local.gateway_otel_env,
local.billing_metrics_env,
local.gateway_timeout_env,
local.gateway_extra_env_list,
local.proxy_config_env,
local.metrics_env,

View file

@ -878,3 +878,14 @@ variable "collector_drain_timeout_seconds" {
error_message = "collector_drain_timeout_seconds must be > 0."
}
}
variable "alb_idle_timeout_seconds" {
description = "ALB idle timeout in seconds. Streaming responses that stay silent longer than this (e.g. a slow first token) are cut by the ALB with a 504, so keep it at or above the proxy's request_timeout."
type = number
default = 600
validation {
condition = var.alb_idle_timeout_seconds >= 1 && var.alb_idle_timeout_seconds <= 4000
error_message = "alb_idle_timeout_seconds must be between 1 and 4000."
}
}

View file

@ -105,6 +105,18 @@ container under `template.template.containers` (Cloud Run v2 supports
multiple containers) and replace the password-based URL with the proxy's
Unix socket.
### Load balancer and gateway timeouts
The external load balancer and gateway Cloud Run service default to 600 seconds
through `lb_timeout_seconds`. The gateway receives `KEEPALIVE_TIMEOUT` set to
30 seconds above that value, so uvicorn keeps connections open longer than the
load balancer. Override `gateway_extra_env.KEEPALIVE_TIMEOUT` when a different
gateway timeout is needed.
```hcl
lb_timeout_seconds = 600
```
## Configuring the proxy
### `proxy_config`

View file

@ -115,6 +115,9 @@ locals {
backend_extra_env_kv = [
for k, v in var.backend_extra_env : { name = k, value = v }
]
gateway_timeout_env_kv = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [
{ name = "KEEPALIVE_TIMEOUT", value = tostring(var.lb_timeout_seconds + 30) },
]
backend_default_env_kv = [
{ name = "STORE_MODEL_IN_DB", value = "true" },
@ -186,7 +189,7 @@ locals {
{ name = "LITELLM_COLLECTOR_DRAIN_TIMEOUT_SECONDS", value = tostring(var.collector_drain_timeout_seconds) },
] : []
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_timeout_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
gateway_env_secrets = concat(local.shared_env_secrets, local.otel_env_secrets, local.billing_metrics_env_secrets, local.gateway_extra_secret_kv)
collector_env_kv_all = concat(
@ -241,6 +244,7 @@ resource "google_cloud_run_v2_service" "gateway" {
template {
service_account = google_service_account.runtime.email
max_instance_request_concurrency = var.gateway_max_instance_request_concurrency
timeout = "${var.lb_timeout_seconds}s"
vpc_access {
connector = google_vpc_access_connector.this[0].id

View file

@ -64,6 +64,7 @@ resource "google_compute_backend_service" "gateway" {
name = "${local.name}-gateway-bs"
protocol = "HTTP"
load_balancing_scheme = "EXTERNAL_MANAGED"
timeout_sec = var.lb_timeout_seconds
backend {
group = google_compute_region_network_endpoint_group.gateway[0].id

View file

@ -726,3 +726,14 @@ variable "collector_drain_timeout_seconds" {
error_message = "collector_drain_timeout_seconds must be > 0."
}
}
variable "lb_timeout_seconds" {
description = "Request timeout in seconds for the external load balancer backend service and the gateway Cloud Run service. Streams that stay silent longer than this (slow first token) are cut with a 504, so keep it at or above the proxy's request_timeout."
type = number
default = 600
validation {
condition = var.lb_timeout_seconds >= 1
error_message = "lb_timeout_seconds must be >= 1."
}
}