mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(infra): default LB and gateway timeouts across GCP and Helm
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
e9ca80c9ba
commit
d0cfcedfb2
13 changed files with 178 additions and 12 deletions
|
|
@ -36,12 +36,14 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
|||
| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` |
|
||||
| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` |
|
||||
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
|
||||
| `keepaliveTimeoutSeconds` | Uvicorn keep-alive timeout in seconds. Must exceed the load balancer idle timeout in front of the gateway. | `630` |
|
||||
| `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` |
|
||||
| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
|
||||
| `ingress.labels` | Additional labels for the Ingress resource | `{}` |
|
||||
| `ingress.idleTimeoutSeconds` | Load balancer idle timeout in seconds. Renders ALB idle-timeout or nginx proxy read/send timeout annotations unless those keys are user-supplied. Set to `0` to disable. | `600` |
|
||||
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
|
||||
| `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` |
|
||||
| `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` |
|
||||
|
|
|
|||
|
|
@ -57,6 +57,10 @@ spec:
|
|||
imagePullPolicy: {{ .Values.image.pullPolicy }}
|
||||
env:
|
||||
{{- include "litellm.proxyEnv" . | nindent 12 }}
|
||||
{{- if and .Values.keepaliveTimeoutSeconds (not (hasKey (default dict .Values.envVars) "KEEPALIVE_TIMEOUT")) }}
|
||||
- name: KEEPALIVE_TIMEOUT
|
||||
value: {{ .Values.keepaliveTimeoutSeconds | quote }}
|
||||
{{- end }}
|
||||
{{- include "litellm.proxyMetricsEnv" . | nindent 12 }}
|
||||
{{- if .Values.collector.enabled }}
|
||||
{{- include "litellm.collectorEnv" . | nindent 12 }}
|
||||
|
|
|
|||
|
|
@ -6,6 +6,17 @@
|
|||
{{- $_ := set .Values.ingress.annotations "kubernetes.io/ingress.class" .Values.ingress.className}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
|
||||
{{- if and (eq .Values.ingress.className "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
|
||||
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- else if and (eq .Values.ingress.className "nginx") .Values.ingress.idleTimeoutSeconds -}}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion -}}
|
||||
apiVersion: networking.k8s.io/v1
|
||||
{{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion -}}
|
||||
|
|
@ -21,9 +32,9 @@ metadata:
|
|||
{{- with .Values.ingress.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
{{- with .Values.ingress.annotations }}
|
||||
{{- if $annotations }}
|
||||
annotations:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- toYaml $annotations | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
{{- if and .Values.ingress.className (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion) }}
|
||||
|
|
|
|||
|
|
@ -16,6 +16,41 @@ tests:
|
|||
- equal:
|
||||
path: spec.template.spec.containers[0].image
|
||||
value: ghcr.io/berriai/litellm:test
|
||||
- it: should set KEEPALIVE_TIMEOUT by default
|
||||
template: deployment.yaml
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "630"
|
||||
- it: should omit KEEPALIVE_TIMEOUT when configured to zero
|
||||
template: deployment.yaml
|
||||
set:
|
||||
keepaliveTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
any: true
|
||||
- it: should preserve an envVars KEEPALIVE_TIMEOUT override without duplication
|
||||
template: deployment.yaml
|
||||
set:
|
||||
envVars:
|
||||
KEEPALIVE_TIMEOUT: "120"
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "120"
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "630"
|
||||
any: true
|
||||
- it: should work with tolerations
|
||||
template: deployment.yaml
|
||||
set:
|
||||
|
|
|
|||
|
|
@ -16,6 +16,49 @@ tests:
|
|||
- isKind:
|
||||
of: Ingress
|
||||
|
||||
- it: should add the default nginx proxy timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "600"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "600"
|
||||
|
||||
- it: should add the default ALB idle timeout annotation
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.className: alb
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=600
|
||||
|
||||
- it: should preserve user-supplied timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.annotations:
|
||||
alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true"
|
||||
ingress.className: alb
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
|
||||
|
||||
- it: should omit timeout annotations when idle timeout is zero
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.idleTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
|
||||
- it: should add custom labels
|
||||
set:
|
||||
ingress.enabled: true
|
||||
|
|
|
|||
|
|
@ -117,6 +117,12 @@ startupProbe:
|
|||
ingress:
|
||||
enabled: false
|
||||
className: "nginx"
|
||||
# Load balancer idle timeout in seconds. ALB renders
|
||||
# alb.ingress.kubernetes.io/load-balancer-attributes:
|
||||
# idle_timeout.timeout_seconds=N. Nginx renders
|
||||
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
|
||||
# User-supplied annotations win, and 0 disables generated annotations.
|
||||
idleTimeoutSeconds: 600
|
||||
labels: {}
|
||||
annotations:
|
||||
{}
|
||||
|
|
@ -586,6 +592,10 @@ migrationJob:
|
|||
# this injection entirely when envVars already defines LITELLM_LOG.
|
||||
logLevel: INFO
|
||||
|
||||
# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the
|
||||
# load balancer idle timeout in front of the gateway.
|
||||
keepaliveTimeoutSeconds: 630
|
||||
|
||||
# Additional environment variables to be added to the deployment as a map of key-value pairs
|
||||
envVars: {}
|
||||
|
||||
|
|
|
|||
|
|
@ -7,8 +7,15 @@
|
|||
{{- $uiPort := .Values.ui.service.port -}}
|
||||
{{- $controller := .Values.ingress.controller | default "alb" -}}
|
||||
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
|
||||
{{- if and (eq $controller "alb") .Values.ingress.albIdleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
|
||||
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.albIdleTimeoutSeconds) -}}
|
||||
{{- if and (eq $controller "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
|
||||
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- else if and (eq $controller "nginx") .Values.ingress.idleTimeoutSeconds -}}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if not (has $controller (list "alb" "nginx")) }}
|
||||
{{- fail (printf "ingress.controller: unknown controller %q, expected one of alb, nginx" $controller) }}
|
||||
|
|
|
|||
|
|
@ -22,21 +22,46 @@ tests:
|
|||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
|
||||
|
||||
- it: does not add the ALB idle timeout annotation for ingress-nginx
|
||||
- it: adds nginx proxy timeout annotations for ingress-nginx
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.controller: nginx
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "600"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "600"
|
||||
|
||||
- it: does not add the ALB idle timeout annotation when disabled
|
||||
- it: preserves user-supplied nginx proxy timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.albIdleTimeoutSeconds: 0
|
||||
ingress.controller: nginx
|
||||
ingress.annotations:
|
||||
nginx.ingress.kubernetes.io/proxy-read-timeout: "120"
|
||||
nginx.ingress.kubernetes.io/proxy-send-timeout: "180"
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "120"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "180"
|
||||
|
||||
- it: does not add load balancer timeout annotations when disabled
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.idleTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
|
||||
- it: keeps the AWS Load Balancer Controller path types by default
|
||||
set:
|
||||
|
|
|
|||
|
|
@ -23,11 +23,12 @@ ingress:
|
|||
# wildcard pathType, so that rule could never match there.
|
||||
controller: alb
|
||||
annotations: {}
|
||||
# ALB idle timeout in seconds, rendered as the
|
||||
# alb.ingress.kubernetes.io/load-balancer-attributes annotation when
|
||||
# controller is alb and ingress.annotations does not set that key itself.
|
||||
# Load balancer idle timeout in seconds. ALB renders
|
||||
# alb.ingress.kubernetes.io/load-balancer-attributes:
|
||||
# idle_timeout.timeout_seconds=N. Nginx renders
|
||||
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
|
||||
# Streams silent longer than this (slow first token) are cut with a 504.
|
||||
albIdleTimeoutSeconds: 600
|
||||
idleTimeoutSeconds: 600
|
||||
host: "" # optional; if set, becomes the rule's host
|
||||
tls: []
|
||||
# Extra HTTP paths appended to the ingress rule. Additive: every built-in
|
||||
|
|
|
|||
|
|
@ -105,6 +105,18 @@ container under `template.template.containers` (Cloud Run v2 supports
|
|||
multiple containers) and replace the password-based URL with the proxy's
|
||||
Unix socket.
|
||||
|
||||
### Load balancer and gateway timeouts
|
||||
|
||||
The external load balancer and gateway Cloud Run service default to 600 seconds
|
||||
through `lb_timeout_seconds`. The gateway receives `KEEPALIVE_TIMEOUT` set to
|
||||
30 seconds above that value, so uvicorn keeps connections open longer than the
|
||||
load balancer. Override `gateway_extra_env.KEEPALIVE_TIMEOUT` when a different
|
||||
gateway timeout is needed.
|
||||
|
||||
```hcl
|
||||
lb_timeout_seconds = 600
|
||||
```
|
||||
|
||||
## Configuring the proxy
|
||||
|
||||
### `proxy_config`
|
||||
|
|
|
|||
|
|
@ -115,6 +115,9 @@ locals {
|
|||
backend_extra_env_kv = [
|
||||
for k, v in var.backend_extra_env : { name = k, value = v }
|
||||
]
|
||||
gateway_timeout_env_kv = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [
|
||||
{ name = "KEEPALIVE_TIMEOUT", value = tostring(var.lb_timeout_seconds + 30) },
|
||||
]
|
||||
|
||||
backend_default_env_kv = [
|
||||
{ name = "STORE_MODEL_IN_DB", value = "true" },
|
||||
|
|
@ -186,7 +189,7 @@ locals {
|
|||
{ name = "LITELLM_COLLECTOR_DRAIN_TIMEOUT_SECONDS", value = tostring(var.collector_drain_timeout_seconds) },
|
||||
] : []
|
||||
|
||||
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
|
||||
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_timeout_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
|
||||
gateway_env_secrets = concat(local.shared_env_secrets, local.otel_env_secrets, local.billing_metrics_env_secrets, local.gateway_extra_secret_kv)
|
||||
|
||||
collector_env_kv_all = concat(
|
||||
|
|
@ -241,6 +244,7 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
template {
|
||||
service_account = google_service_account.runtime.email
|
||||
max_instance_request_concurrency = var.gateway_max_instance_request_concurrency
|
||||
timeout = "${var.lb_timeout_seconds}s"
|
||||
|
||||
vpc_access {
|
||||
connector = google_vpc_access_connector.this[0].id
|
||||
|
|
|
|||
|
|
@ -64,6 +64,7 @@ resource "google_compute_backend_service" "gateway" {
|
|||
name = "${local.name}-gateway-bs"
|
||||
protocol = "HTTP"
|
||||
load_balancing_scheme = "EXTERNAL_MANAGED"
|
||||
timeout_sec = var.lb_timeout_seconds
|
||||
|
||||
backend {
|
||||
group = google_compute_region_network_endpoint_group.gateway[0].id
|
||||
|
|
|
|||
|
|
@ -726,3 +726,14 @@ variable "collector_drain_timeout_seconds" {
|
|||
error_message = "collector_drain_timeout_seconds must be > 0."
|
||||
}
|
||||
}
|
||||
|
||||
variable "lb_timeout_seconds" {
|
||||
description = "Request timeout in seconds for the external load balancer backend service and the gateway Cloud Run service. Streams that stay silent longer than this (slow first token) are cut with a 504, so keep it at or above the proxy's request_timeout."
|
||||
type = number
|
||||
default = 600
|
||||
|
||||
validation {
|
||||
condition = var.lb_timeout_seconds >= 1
|
||||
error_message = "lb_timeout_seconds must be >= 1."
|
||||
}
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue