diff --git a/helm/litellm-helm/README.md b/helm/litellm-helm/README.md index bf4089404db..24c3d84e437 100644 --- a/helm/litellm-helm/README.md +++ b/helm/litellm-helm/README.md @@ -36,12 +36,14 @@ If `db.useStackgresOperator` is used (not yet implemented): | `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` | | `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` | | `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` | +| `keepaliveTimeoutSeconds` | Uvicorn keep-alive timeout in seconds. Must exceed the load balancer idle timeout in front of the gateway. | `630` | | `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | | `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | | `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | | `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` | | `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` | | `ingress.labels` | Additional labels for the Ingress resource | `{}` | +| `ingress.idleTimeoutSeconds` | Load balancer idle timeout in seconds. Renders ALB idle-timeout or nginx proxy read/send timeout annotations unless those keys are user-supplied. Set to `0` to disable. | `600` | | `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A | | `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` | | `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` | diff --git a/helm/litellm-helm/templates/deployment.yaml b/helm/litellm-helm/templates/deployment.yaml index cf7b3f8a38d..8277e7eac9e 100644 --- a/helm/litellm-helm/templates/deployment.yaml +++ b/helm/litellm-helm/templates/deployment.yaml @@ -57,6 +57,10 @@ spec: imagePullPolicy: {{ .Values.image.pullPolicy }} env: {{- include "litellm.proxyEnv" . | nindent 12 }} + {{- if and .Values.keepaliveTimeoutSeconds (not (hasKey (default dict .Values.envVars) "KEEPALIVE_TIMEOUT")) }} + - name: KEEPALIVE_TIMEOUT + value: {{ .Values.keepaliveTimeoutSeconds | quote }} + {{- end }} {{- include "litellm.proxyMetricsEnv" . | nindent 12 }} {{- if .Values.collector.enabled }} {{- include "litellm.collectorEnv" . | nindent 12 }} diff --git a/helm/litellm-helm/templates/ingress.yaml b/helm/litellm-helm/templates/ingress.yaml index ea9ffcbb54c..3610d66e4e4 100644 --- a/helm/litellm-helm/templates/ingress.yaml +++ b/helm/litellm-helm/templates/ingress.yaml @@ -6,6 +6,17 @@ {{- $_ := set .Values.ingress.annotations "kubernetes.io/ingress.class" .Values.ingress.className}} {{- end }} {{- end }} +{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}} +{{- if and (eq .Values.ingress.className "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}} +{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- else if and (eq .Values.ingress.className "nginx") .Values.ingress.idleTimeoutSeconds -}} +{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}} +{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- end }} +{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}} +{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- end }} +{{- end }} {{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion -}} apiVersion: networking.k8s.io/v1 {{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion -}} @@ -21,9 +32,9 @@ metadata: {{- with .Values.ingress.labels }} {{- toYaml . | nindent 4 }} {{- end }} - {{- with .Values.ingress.annotations }} + {{- if $annotations }} annotations: - {{- toYaml . | nindent 4 }} + {{- toYaml $annotations | nindent 4 }} {{- end }} spec: {{- if and .Values.ingress.className (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion) }} diff --git a/helm/litellm-helm/tests/deployment_tests.yaml b/helm/litellm-helm/tests/deployment_tests.yaml index ee946038202..858e3a38252 100644 --- a/helm/litellm-helm/tests/deployment_tests.yaml +++ b/helm/litellm-helm/tests/deployment_tests.yaml @@ -16,6 +16,41 @@ tests: - equal: path: spec.template.spec.containers[0].image value: ghcr.io/berriai/litellm:test + - it: should set KEEPALIVE_TIMEOUT by default + template: deployment.yaml + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + value: "630" + - it: should omit KEEPALIVE_TIMEOUT when configured to zero + template: deployment.yaml + set: + keepaliveTimeoutSeconds: 0 + asserts: + - notContains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + any: true + - it: should preserve an envVars KEEPALIVE_TIMEOUT override without duplication + template: deployment.yaml + set: + envVars: + KEEPALIVE_TIMEOUT: "120" + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + value: "120" + - notContains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + value: "630" + any: true - it: should work with tolerations template: deployment.yaml set: diff --git a/helm/litellm-helm/tests/ingress_tests.yaml b/helm/litellm-helm/tests/ingress_tests.yaml index aad6ecfcee8..f8228ea4208 100644 --- a/helm/litellm-helm/tests/ingress_tests.yaml +++ b/helm/litellm-helm/tests/ingress_tests.yaml @@ -16,6 +16,49 @@ tests: - isKind: of: Ingress + - it: should add the default nginx proxy timeout annotations + set: + ingress.enabled: true + asserts: + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"] + value: "600" + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"] + value: "600" + + - it: should add the default ALB idle timeout annotation + set: + ingress.enabled: true + ingress.className: alb + asserts: + - equal: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + value: idle_timeout.timeout_seconds=600 + + - it: should preserve user-supplied timeout annotations + set: + ingress.enabled: true + ingress.annotations: + alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true" + ingress.className: alb + asserts: + - equal: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true + + - it: should omit timeout annotations when idle timeout is zero + set: + ingress.enabled: true + ingress.idleTimeoutSeconds: 0 + asserts: + - notExists: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + - notExists: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"] + - notExists: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"] + - it: should add custom labels set: ingress.enabled: true diff --git a/helm/litellm-helm/values.yaml b/helm/litellm-helm/values.yaml index fcee331a5aa..a61d36f2a9a 100644 --- a/helm/litellm-helm/values.yaml +++ b/helm/litellm-helm/values.yaml @@ -117,6 +117,12 @@ startupProbe: ingress: enabled: false className: "nginx" + # Load balancer idle timeout in seconds. ALB renders + # alb.ingress.kubernetes.io/load-balancer-attributes: + # idle_timeout.timeout_seconds=N. Nginx renders + # nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout. + # User-supplied annotations win, and 0 disables generated annotations. + idleTimeoutSeconds: 600 labels: {} annotations: {} @@ -586,6 +592,10 @@ migrationJob: # this injection entirely when envVars already defines LITELLM_LOG. logLevel: INFO +# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the +# load balancer idle timeout in front of the gateway. +keepaliveTimeoutSeconds: 630 + # Additional environment variables to be added to the deployment as a map of key-value pairs envVars: {} diff --git a/helm/litellm/templates/gateway/deployment.yaml b/helm/litellm/templates/gateway/deployment.yaml index 49b452b3053..f2fb169891f 100644 --- a/helm/litellm/templates/gateway/deployment.yaml +++ b/helm/litellm/templates/gateway/deployment.yaml @@ -64,6 +64,10 @@ spec: - name: NUM_WORKERS value: {{ .Values.gateway.numWorkers | quote }} {{- end }} + {{- if .Values.gateway.keepaliveTimeoutSeconds }} + - name: KEEPALIVE_TIMEOUT + value: {{ .Values.gateway.keepaliveTimeoutSeconds | quote }} + {{- end }} {{- if .Values.database.connectionPool.enabled }} {{- include "litellm.connectionPoolEnv" $ | nindent 12 }} {{- end }} diff --git a/helm/litellm/templates/ingress.yaml b/helm/litellm/templates/ingress.yaml index e9f7ed4ec3f..8a6e884b7fc 100644 --- a/helm/litellm/templates/ingress.yaml +++ b/helm/litellm/templates/ingress.yaml @@ -6,6 +6,17 @@ {{- $backendPort := .Values.backend.service.port -}} {{- $uiPort := .Values.ui.service.port -}} {{- $controller := .Values.ingress.controller | default "alb" -}} +{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}} +{{- if and (eq $controller "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}} +{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- else if and (eq $controller "nginx") .Values.ingress.idleTimeoutSeconds -}} +{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}} +{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- end }} +{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}} +{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}} +{{- end }} +{{- end }} {{- if not (has $controller (list "alb" "nginx")) }} {{- fail (printf "ingress.controller: unknown controller %q, expected one of alb, nginx" $controller) }} {{- end }} @@ -96,9 +107,9 @@ metadata: name: {{ include "litellm.fullname" . }} labels: {{- include "litellm.commonLabels" . | nindent 4 }} - {{- with .Values.ingress.annotations }} + {{- if $annotations }} annotations: - {{- toYaml . | nindent 4 }} + {{- toYaml $annotations | nindent 4 }} {{- end }} spec: {{- with .Values.ingress.className }} diff --git a/helm/litellm/tests/gateway_env_tests.yaml b/helm/litellm/tests/gateway_env_tests.yaml new file mode 100644 index 00000000000..85620e00dbc --- /dev/null +++ b/helm/litellm/tests/gateway_env_tests.yaml @@ -0,0 +1,33 @@ +suite: test gateway environment defaults +templates: + - gateway/deployment.yaml + - gateway/configmap.yaml +values: + - ./values/required.yaml +tests: + - it: sets KEEPALIVE_TIMEOUT only on the gateway container + template: gateway/deployment.yaml + set: + gateway.collector.enabled: true + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + value: "630" + - notContains: + path: spec.template.spec.containers[1].env + content: + name: KEEPALIVE_TIMEOUT + any: true + + - it: omits KEEPALIVE_TIMEOUT when configured to zero + template: gateway/deployment.yaml + set: + gateway.keepaliveTimeoutSeconds: 0 + asserts: + - notContains: + path: spec.template.spec.containers[0].env + content: + name: KEEPALIVE_TIMEOUT + any: true diff --git a/helm/litellm/tests/ingress_controller_tests.yaml b/helm/litellm/tests/ingress_controller_tests.yaml index aa30db3c9c1..b0af1dc0cb0 100644 --- a/helm/litellm/tests/ingress_controller_tests.yaml +++ b/helm/litellm/tests/ingress_controller_tests.yaml @@ -4,6 +4,65 @@ templates: values: - ./values/required.yaml tests: + - it: adds the default ALB idle timeout annotation + set: + ingress.enabled: true + asserts: + - equal: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + value: idle_timeout.timeout_seconds=600 + + - it: preserves a user-supplied ALB load balancer attributes annotation + set: + ingress.enabled: true + ingress.annotations: + alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true" + asserts: + - equal: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true + + - it: adds nginx proxy timeout annotations for ingress-nginx + set: + ingress.enabled: true + ingress.controller: nginx + asserts: + - notExists: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"] + value: "600" + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"] + value: "600" + + - it: preserves user-supplied nginx proxy timeout annotations + set: + ingress.enabled: true + ingress.controller: nginx + ingress.annotations: + nginx.ingress.kubernetes.io/proxy-read-timeout: "120" + nginx.ingress.kubernetes.io/proxy-send-timeout: "180" + asserts: + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"] + value: "120" + - equal: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"] + value: "180" + + - it: does not add load balancer timeout annotations when disabled + set: + ingress.enabled: true + ingress.idleTimeoutSeconds: 0 + asserts: + - notExists: + path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"] + - notExists: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"] + - notExists: + path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"] + - it: keeps the AWS Load Balancer Controller path types by default set: ingress.enabled: true diff --git a/helm/litellm/values.yaml b/helm/litellm/values.yaml index 2c0c7151a32..1238050db6a 100644 --- a/helm/litellm/values.yaml +++ b/helm/litellm/values.yaml @@ -23,6 +23,12 @@ ingress: # wildcard pathType, so that rule could never match there. controller: alb annotations: {} + # Load balancer idle timeout in seconds. ALB renders + # alb.ingress.kubernetes.io/load-balancer-attributes: + # idle_timeout.timeout_seconds=N. Nginx renders + # nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout. + # Streams silent longer than this (slow first token) are cut with a 504. + idleTimeoutSeconds: 600 host: "" # optional; if set, becomes the rule's host tls: [] # Extra HTTP paths appended to the ingress rule. Additive: every built-in @@ -282,6 +288,10 @@ gateway: # Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS, # consumed by the gateway image entrypoint. Default is 1. numWorkers: 1 + # Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the + # load balancer idle timeout in front of the gateway, otherwise the LB can + # hand a request to a connection uvicorn is closing. + keepaliveTimeoutSeconds: 630 extraEnv: [] # Add extra environment variables to the gateway envConfigMaps: [] # Add extra environment variables to the gateway from config maps envSecrets: [] # Add extra environment variables to the gateway from secrets diff --git a/terraform/litellm/aws/README.md b/terraform/litellm/aws/README.md index 6fcbdc2f500..66305b11c05 100644 --- a/terraform/litellm/aws/README.md +++ b/terraform/litellm/aws/README.md @@ -335,6 +335,17 @@ gateway_tokens_metric = { } ``` +### Load balancer and gateway timeouts + +The ALB idle timeout defaults to 600 seconds through `alb_idle_timeout_seconds`. +The gateway receives `KEEPALIVE_TIMEOUT` set to 30 seconds above that value, so +uvicorn keeps connections open longer than the load balancer. Override +`gateway_extra_env.KEEPALIVE_TIMEOUT` when a different gateway timeout is needed. + +```hcl +alb_idle_timeout_seconds = 600 +``` + Worked example for the request policy: 1,000 rps across 10 tasks is 100 rps per task (the ALB reports it as 6,000 per minute per target) against a target of 90 (5,400), so target tracking sizes the service to diff --git a/terraform/litellm/aws/alb.tf b/terraform/litellm/aws/alb.tf index bb07a83caa7..b17ff89a25c 100644 --- a/terraform/litellm/aws/alb.tf +++ b/terraform/litellm/aws/alb.tf @@ -5,7 +5,7 @@ resource "aws_lb" "this" { security_groups = [aws_security_group.alb.id] subnets = local.public_subnet_ids - idle_timeout = 120 + idle_timeout = var.alb_idle_timeout_seconds lifecycle { precondition { diff --git a/terraform/litellm/aws/ecs.tf b/terraform/litellm/aws/ecs.tf index 2b235c2bad5..a26bb97dd23 100644 --- a/terraform/litellm/aws/ecs.tf +++ b/terraform/litellm/aws/ecs.tf @@ -176,6 +176,9 @@ locals { backend_extra_env_list = [ for k, v in var.backend_extra_env : { name = k, value = v } ] + gateway_timeout_env = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [ + { name = "KEEPALIVE_TIMEOUT", value = tostring(var.alb_idle_timeout_seconds + 30) }, + ] # Storing models in the DB needs a DB. Without one the backend reads its # model list from proxy_config only. @@ -292,6 +295,7 @@ locals { local.shared_env, local.gateway_otel_env, local.billing_metrics_env, + local.gateway_timeout_env, local.gateway_extra_env_list, local.proxy_config_env, local.metrics_env, diff --git a/terraform/litellm/aws/variables.tf b/terraform/litellm/aws/variables.tf index 580a0cc657a..b4851b8ca46 100644 --- a/terraform/litellm/aws/variables.tf +++ b/terraform/litellm/aws/variables.tf @@ -878,3 +878,14 @@ variable "collector_drain_timeout_seconds" { error_message = "collector_drain_timeout_seconds must be > 0." } } + +variable "alb_idle_timeout_seconds" { + description = "ALB idle timeout in seconds. Streaming responses that stay silent longer than this (e.g. a slow first token) are cut by the ALB with a 504, so keep it at or above the proxy's request_timeout." + type = number + default = 600 + + validation { + condition = var.alb_idle_timeout_seconds >= 1 && var.alb_idle_timeout_seconds <= 4000 + error_message = "alb_idle_timeout_seconds must be between 1 and 4000." + } +} diff --git a/terraform/litellm/gcp/README.md b/terraform/litellm/gcp/README.md index 4b2f576adc1..5c692f55194 100644 --- a/terraform/litellm/gcp/README.md +++ b/terraform/litellm/gcp/README.md @@ -105,6 +105,18 @@ container under `template.template.containers` (Cloud Run v2 supports multiple containers) and replace the password-based URL with the proxy's Unix socket. +### Load balancer and gateway timeouts + +The external load balancer and gateway Cloud Run service default to 600 seconds +through `lb_timeout_seconds`. The gateway receives `KEEPALIVE_TIMEOUT` set to +30 seconds above that value, so uvicorn keeps connections open longer than the +load balancer. Override `gateway_extra_env.KEEPALIVE_TIMEOUT` when a different +gateway timeout is needed. + +```hcl +lb_timeout_seconds = 600 +``` + ## Configuring the proxy ### `proxy_config` diff --git a/terraform/litellm/gcp/cloudrun.tf b/terraform/litellm/gcp/cloudrun.tf index d0b32a367d6..3a96286e39d 100644 --- a/terraform/litellm/gcp/cloudrun.tf +++ b/terraform/litellm/gcp/cloudrun.tf @@ -115,6 +115,9 @@ locals { backend_extra_env_kv = [ for k, v in var.backend_extra_env : { name = k, value = v } ] + gateway_timeout_env_kv = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [ + { name = "KEEPALIVE_TIMEOUT", value = tostring(var.lb_timeout_seconds + 30) }, + ] backend_default_env_kv = [ { name = "STORE_MODEL_IN_DB", value = "true" }, @@ -186,7 +189,7 @@ locals { { name = "LITELLM_COLLECTOR_DRAIN_TIMEOUT_SECONDS", value = tostring(var.collector_drain_timeout_seconds) }, ] : [] - gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv) + gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_timeout_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv) gateway_env_secrets = concat(local.shared_env_secrets, local.otel_env_secrets, local.billing_metrics_env_secrets, local.gateway_extra_secret_kv) collector_env_kv_all = concat( @@ -241,6 +244,7 @@ resource "google_cloud_run_v2_service" "gateway" { template { service_account = google_service_account.runtime.email max_instance_request_concurrency = var.gateway_max_instance_request_concurrency + timeout = "${var.lb_timeout_seconds}s" vpc_access { connector = google_vpc_access_connector.this[0].id diff --git a/terraform/litellm/gcp/load_balancer.tf b/terraform/litellm/gcp/load_balancer.tf index 57e8af8210f..14913ba1e0d 100644 --- a/terraform/litellm/gcp/load_balancer.tf +++ b/terraform/litellm/gcp/load_balancer.tf @@ -64,6 +64,7 @@ resource "google_compute_backend_service" "gateway" { name = "${local.name}-gateway-bs" protocol = "HTTP" load_balancing_scheme = "EXTERNAL_MANAGED" + timeout_sec = var.lb_timeout_seconds backend { group = google_compute_region_network_endpoint_group.gateway[0].id diff --git a/terraform/litellm/gcp/variables.tf b/terraform/litellm/gcp/variables.tf index 412f919ab89..2e5c195c9c2 100644 --- a/terraform/litellm/gcp/variables.tf +++ b/terraform/litellm/gcp/variables.tf @@ -726,3 +726,14 @@ variable "collector_drain_timeout_seconds" { error_message = "collector_drain_timeout_seconds must be > 0." } } + +variable "lb_timeout_seconds" { + description = "Request timeout in seconds for the external load balancer backend service and the gateway Cloud Run service. Streams that stay silent longer than this (slow first token) are cut with a 504, so keep it at or above the proxy's request_timeout." + type = number + default = 600 + + validation { + condition = var.lb_timeout_seconds >= 1 + error_message = "lb_timeout_seconds must be >= 1." + } +}