mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Merge d0cfcedfb2 into b79fc9f1b0
This commit is contained in:
commit
87c31c86f0
19 changed files with 282 additions and 6 deletions
|
|
@ -36,12 +36,14 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
|||
| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` |
|
||||
| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` |
|
||||
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
|
||||
| `keepaliveTimeoutSeconds` | Uvicorn keep-alive timeout in seconds. Must exceed the load balancer idle timeout in front of the gateway. | `630` |
|
||||
| `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` |
|
||||
| `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` |
|
||||
| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
|
||||
| `ingress.labels` | Additional labels for the Ingress resource | `{}` |
|
||||
| `ingress.idleTimeoutSeconds` | Load balancer idle timeout in seconds. Renders ALB idle-timeout or nginx proxy read/send timeout annotations unless those keys are user-supplied. Set to `0` to disable. | `600` |
|
||||
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
|
||||
| `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` |
|
||||
| `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` |
|
||||
|
|
|
|||
|
|
@ -57,6 +57,10 @@ spec:
|
|||
imagePullPolicy: {{ .Values.image.pullPolicy }}
|
||||
env:
|
||||
{{- include "litellm.proxyEnv" . | nindent 12 }}
|
||||
{{- if and .Values.keepaliveTimeoutSeconds (not (hasKey (default dict .Values.envVars) "KEEPALIVE_TIMEOUT")) }}
|
||||
- name: KEEPALIVE_TIMEOUT
|
||||
value: {{ .Values.keepaliveTimeoutSeconds | quote }}
|
||||
{{- end }}
|
||||
{{- include "litellm.proxyMetricsEnv" . | nindent 12 }}
|
||||
{{- if .Values.collector.enabled }}
|
||||
{{- include "litellm.collectorEnv" . | nindent 12 }}
|
||||
|
|
|
|||
|
|
@ -6,6 +6,17 @@
|
|||
{{- $_ := set .Values.ingress.annotations "kubernetes.io/ingress.class" .Values.ingress.className}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
|
||||
{{- if and (eq .Values.ingress.className "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
|
||||
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- else if and (eq .Values.ingress.className "nginx") .Values.ingress.idleTimeoutSeconds -}}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion -}}
|
||||
apiVersion: networking.k8s.io/v1
|
||||
{{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion -}}
|
||||
|
|
@ -21,9 +32,9 @@ metadata:
|
|||
{{- with .Values.ingress.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
{{- with .Values.ingress.annotations }}
|
||||
{{- if $annotations }}
|
||||
annotations:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- toYaml $annotations | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
{{- if and .Values.ingress.className (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion) }}
|
||||
|
|
|
|||
|
|
@ -16,6 +16,41 @@ tests:
|
|||
- equal:
|
||||
path: spec.template.spec.containers[0].image
|
||||
value: ghcr.io/berriai/litellm:test
|
||||
- it: should set KEEPALIVE_TIMEOUT by default
|
||||
template: deployment.yaml
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "630"
|
||||
- it: should omit KEEPALIVE_TIMEOUT when configured to zero
|
||||
template: deployment.yaml
|
||||
set:
|
||||
keepaliveTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
any: true
|
||||
- it: should preserve an envVars KEEPALIVE_TIMEOUT override without duplication
|
||||
template: deployment.yaml
|
||||
set:
|
||||
envVars:
|
||||
KEEPALIVE_TIMEOUT: "120"
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "120"
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "630"
|
||||
any: true
|
||||
- it: should work with tolerations
|
||||
template: deployment.yaml
|
||||
set:
|
||||
|
|
|
|||
|
|
@ -16,6 +16,49 @@ tests:
|
|||
- isKind:
|
||||
of: Ingress
|
||||
|
||||
- it: should add the default nginx proxy timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "600"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "600"
|
||||
|
||||
- it: should add the default ALB idle timeout annotation
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.className: alb
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=600
|
||||
|
||||
- it: should preserve user-supplied timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.annotations:
|
||||
alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true"
|
||||
ingress.className: alb
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
|
||||
|
||||
- it: should omit timeout annotations when idle timeout is zero
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.idleTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
|
||||
- it: should add custom labels
|
||||
set:
|
||||
ingress.enabled: true
|
||||
|
|
|
|||
|
|
@ -117,6 +117,12 @@ startupProbe:
|
|||
ingress:
|
||||
enabled: false
|
||||
className: "nginx"
|
||||
# Load balancer idle timeout in seconds. ALB renders
|
||||
# alb.ingress.kubernetes.io/load-balancer-attributes:
|
||||
# idle_timeout.timeout_seconds=N. Nginx renders
|
||||
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
|
||||
# User-supplied annotations win, and 0 disables generated annotations.
|
||||
idleTimeoutSeconds: 600
|
||||
labels: {}
|
||||
annotations:
|
||||
{}
|
||||
|
|
@ -586,6 +592,10 @@ migrationJob:
|
|||
# this injection entirely when envVars already defines LITELLM_LOG.
|
||||
logLevel: INFO
|
||||
|
||||
# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the
|
||||
# load balancer idle timeout in front of the gateway.
|
||||
keepaliveTimeoutSeconds: 630
|
||||
|
||||
# Additional environment variables to be added to the deployment as a map of key-value pairs
|
||||
envVars: {}
|
||||
|
||||
|
|
|
|||
|
|
@ -64,6 +64,10 @@ spec:
|
|||
- name: NUM_WORKERS
|
||||
value: {{ .Values.gateway.numWorkers | quote }}
|
||||
{{- end }}
|
||||
{{- if .Values.gateway.keepaliveTimeoutSeconds }}
|
||||
- name: KEEPALIVE_TIMEOUT
|
||||
value: {{ .Values.gateway.keepaliveTimeoutSeconds | quote }}
|
||||
{{- end }}
|
||||
{{- if .Values.database.connectionPool.enabled }}
|
||||
{{- include "litellm.connectionPoolEnv" $ | nindent 12 }}
|
||||
{{- end }}
|
||||
|
|
|
|||
|
|
@ -6,6 +6,17 @@
|
|||
{{- $backendPort := .Values.backend.service.port -}}
|
||||
{{- $uiPort := .Values.ui.service.port -}}
|
||||
{{- $controller := .Values.ingress.controller | default "alb" -}}
|
||||
{{- $annotations := deepCopy (.Values.ingress.annotations | default (dict)) -}}
|
||||
{{- if and (eq $controller "alb") .Values.ingress.idleTimeoutSeconds (not (hasKey $annotations "alb.ingress.kubernetes.io/load-balancer-attributes")) -}}
|
||||
{{- $_ := set $annotations "alb.ingress.kubernetes.io/load-balancer-attributes" (printf "idle_timeout.timeout_seconds=%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- else if and (eq $controller "nginx") .Values.ingress.idleTimeoutSeconds -}}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-read-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- if not (hasKey $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout") -}}
|
||||
{{- $_ := set $annotations "nginx.ingress.kubernetes.io/proxy-send-timeout" (printf "%v" .Values.ingress.idleTimeoutSeconds) -}}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if not (has $controller (list "alb" "nginx")) }}
|
||||
{{- fail (printf "ingress.controller: unknown controller %q, expected one of alb, nginx" $controller) }}
|
||||
{{- end }}
|
||||
|
|
@ -96,9 +107,9 @@ metadata:
|
|||
name: {{ include "litellm.fullname" . }}
|
||||
labels:
|
||||
{{- include "litellm.commonLabels" . | nindent 4 }}
|
||||
{{- with .Values.ingress.annotations }}
|
||||
{{- if $annotations }}
|
||||
annotations:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- toYaml $annotations | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
{{- with .Values.ingress.className }}
|
||||
|
|
|
|||
33
helm/litellm/tests/gateway_env_tests.yaml
Normal file
33
helm/litellm/tests/gateway_env_tests.yaml
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
suite: test gateway environment defaults
|
||||
templates:
|
||||
- gateway/deployment.yaml
|
||||
- gateway/configmap.yaml
|
||||
values:
|
||||
- ./values/required.yaml
|
||||
tests:
|
||||
- it: sets KEEPALIVE_TIMEOUT only on the gateway container
|
||||
template: gateway/deployment.yaml
|
||||
set:
|
||||
gateway.collector.enabled: true
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
value: "630"
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[1].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
any: true
|
||||
|
||||
- it: omits KEEPALIVE_TIMEOUT when configured to zero
|
||||
template: gateway/deployment.yaml
|
||||
set:
|
||||
gateway.keepaliveTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: KEEPALIVE_TIMEOUT
|
||||
any: true
|
||||
|
|
@ -4,6 +4,65 @@ templates:
|
|||
values:
|
||||
- ./values/required.yaml
|
||||
tests:
|
||||
- it: adds the default ALB idle timeout annotation
|
||||
set:
|
||||
ingress.enabled: true
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=600
|
||||
|
||||
- it: preserves a user-supplied ALB load balancer attributes annotation
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.annotations:
|
||||
alb.ingress.kubernetes.io/load-balancer-attributes: "idle_timeout.timeout_seconds=120,routing.http2.enabled=true"
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
value: idle_timeout.timeout_seconds=120,routing.http2.enabled=true
|
||||
|
||||
- it: adds nginx proxy timeout annotations for ingress-nginx
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.controller: nginx
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "600"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "600"
|
||||
|
||||
- it: preserves user-supplied nginx proxy timeout annotations
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.controller: nginx
|
||||
ingress.annotations:
|
||||
nginx.ingress.kubernetes.io/proxy-read-timeout: "120"
|
||||
nginx.ingress.kubernetes.io/proxy-send-timeout: "180"
|
||||
asserts:
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
value: "120"
|
||||
- equal:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
value: "180"
|
||||
|
||||
- it: does not add load balancer timeout annotations when disabled
|
||||
set:
|
||||
ingress.enabled: true
|
||||
ingress.idleTimeoutSeconds: 0
|
||||
asserts:
|
||||
- notExists:
|
||||
path: metadata.annotations["alb.ingress.kubernetes.io/load-balancer-attributes"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-read-timeout"]
|
||||
- notExists:
|
||||
path: metadata.annotations["nginx.ingress.kubernetes.io/proxy-send-timeout"]
|
||||
|
||||
- it: keeps the AWS Load Balancer Controller path types by default
|
||||
set:
|
||||
ingress.enabled: true
|
||||
|
|
|
|||
|
|
@ -23,6 +23,12 @@ ingress:
|
|||
# wildcard pathType, so that rule could never match there.
|
||||
controller: alb
|
||||
annotations: {}
|
||||
# Load balancer idle timeout in seconds. ALB renders
|
||||
# alb.ingress.kubernetes.io/load-balancer-attributes:
|
||||
# idle_timeout.timeout_seconds=N. Nginx renders
|
||||
# nginx.ingress.kubernetes.io/proxy-read-timeout and proxy-send-timeout.
|
||||
# Streams silent longer than this (slow first token) are cut with a 504.
|
||||
idleTimeoutSeconds: 600
|
||||
host: "" # optional; if set, becomes the rule's host
|
||||
tls: []
|
||||
# Extra HTTP paths appended to the ingress rule. Additive: every built-in
|
||||
|
|
@ -282,6 +288,10 @@ gateway:
|
|||
# Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS,
|
||||
# consumed by the gateway image entrypoint. Default is 1.
|
||||
numWorkers: 1
|
||||
# Uvicorn keep-alive timeout in seconds (KEEPALIVE_TIMEOUT). Must exceed the
|
||||
# load balancer idle timeout in front of the gateway, otherwise the LB can
|
||||
# hand a request to a connection uvicorn is closing.
|
||||
keepaliveTimeoutSeconds: 630
|
||||
extraEnv: [] # Add extra environment variables to the gateway
|
||||
envConfigMaps: [] # Add extra environment variables to the gateway from config maps
|
||||
envSecrets: [] # Add extra environment variables to the gateway from secrets
|
||||
|
|
|
|||
|
|
@ -335,6 +335,17 @@ gateway_tokens_metric = {
|
|||
}
|
||||
```
|
||||
|
||||
### Load balancer and gateway timeouts
|
||||
|
||||
The ALB idle timeout defaults to 600 seconds through `alb_idle_timeout_seconds`.
|
||||
The gateway receives `KEEPALIVE_TIMEOUT` set to 30 seconds above that value, so
|
||||
uvicorn keeps connections open longer than the load balancer. Override
|
||||
`gateway_extra_env.KEEPALIVE_TIMEOUT` when a different gateway timeout is needed.
|
||||
|
||||
```hcl
|
||||
alb_idle_timeout_seconds = 600
|
||||
```
|
||||
|
||||
Worked example for the request policy: 1,000 rps across 10 tasks is 100 rps
|
||||
per task (the ALB reports it as 6,000 per minute per target) against a target
|
||||
of 90 (5,400), so target tracking sizes the service to
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ resource "aws_lb" "this" {
|
|||
security_groups = [aws_security_group.alb.id]
|
||||
subnets = local.public_subnet_ids
|
||||
|
||||
idle_timeout = 120
|
||||
idle_timeout = var.alb_idle_timeout_seconds
|
||||
|
||||
lifecycle {
|
||||
precondition {
|
||||
|
|
|
|||
|
|
@ -176,6 +176,9 @@ locals {
|
|||
backend_extra_env_list = [
|
||||
for k, v in var.backend_extra_env : { name = k, value = v }
|
||||
]
|
||||
gateway_timeout_env = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [
|
||||
{ name = "KEEPALIVE_TIMEOUT", value = tostring(var.alb_idle_timeout_seconds + 30) },
|
||||
]
|
||||
|
||||
# Storing models in the DB needs a DB. Without one the backend reads its
|
||||
# model list from proxy_config only.
|
||||
|
|
@ -292,6 +295,7 @@ locals {
|
|||
local.shared_env,
|
||||
local.gateway_otel_env,
|
||||
local.billing_metrics_env,
|
||||
local.gateway_timeout_env,
|
||||
local.gateway_extra_env_list,
|
||||
local.proxy_config_env,
|
||||
local.metrics_env,
|
||||
|
|
|
|||
|
|
@ -878,3 +878,14 @@ variable "collector_drain_timeout_seconds" {
|
|||
error_message = "collector_drain_timeout_seconds must be > 0."
|
||||
}
|
||||
}
|
||||
|
||||
variable "alb_idle_timeout_seconds" {
|
||||
description = "ALB idle timeout in seconds. Streaming responses that stay silent longer than this (e.g. a slow first token) are cut by the ALB with a 504, so keep it at or above the proxy's request_timeout."
|
||||
type = number
|
||||
default = 600
|
||||
|
||||
validation {
|
||||
condition = var.alb_idle_timeout_seconds >= 1 && var.alb_idle_timeout_seconds <= 4000
|
||||
error_message = "alb_idle_timeout_seconds must be between 1 and 4000."
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -105,6 +105,18 @@ container under `template.template.containers` (Cloud Run v2 supports
|
|||
multiple containers) and replace the password-based URL with the proxy's
|
||||
Unix socket.
|
||||
|
||||
### Load balancer and gateway timeouts
|
||||
|
||||
The external load balancer and gateway Cloud Run service default to 600 seconds
|
||||
through `lb_timeout_seconds`. The gateway receives `KEEPALIVE_TIMEOUT` set to
|
||||
30 seconds above that value, so uvicorn keeps connections open longer than the
|
||||
load balancer. Override `gateway_extra_env.KEEPALIVE_TIMEOUT` when a different
|
||||
gateway timeout is needed.
|
||||
|
||||
```hcl
|
||||
lb_timeout_seconds = 600
|
||||
```
|
||||
|
||||
## Configuring the proxy
|
||||
|
||||
### `proxy_config`
|
||||
|
|
|
|||
|
|
@ -115,6 +115,9 @@ locals {
|
|||
backend_extra_env_kv = [
|
||||
for k, v in var.backend_extra_env : { name = k, value = v }
|
||||
]
|
||||
gateway_timeout_env_kv = contains(keys(var.gateway_extra_env), "KEEPALIVE_TIMEOUT") ? [] : [
|
||||
{ name = "KEEPALIVE_TIMEOUT", value = tostring(var.lb_timeout_seconds + 30) },
|
||||
]
|
||||
|
||||
backend_default_env_kv = [
|
||||
{ name = "STORE_MODEL_IN_DB", value = "true" },
|
||||
|
|
@ -186,7 +189,7 @@ locals {
|
|||
{ name = "LITELLM_COLLECTOR_DRAIN_TIMEOUT_SECONDS", value = tostring(var.collector_drain_timeout_seconds) },
|
||||
] : []
|
||||
|
||||
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
|
||||
gateway_env_kv = concat(local.shared_env_kv, local.gateway_otel_env_kv, local.billing_metrics_env_kv, local.gateway_timeout_env_kv, local.gateway_extra_env_kv, local.proxy_config_env, local.metrics_env_kv, local.gateway_pool_env, local.collector_env_kv)
|
||||
gateway_env_secrets = concat(local.shared_env_secrets, local.otel_env_secrets, local.billing_metrics_env_secrets, local.gateway_extra_secret_kv)
|
||||
|
||||
collector_env_kv_all = concat(
|
||||
|
|
@ -241,6 +244,7 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
template {
|
||||
service_account = google_service_account.runtime.email
|
||||
max_instance_request_concurrency = var.gateway_max_instance_request_concurrency
|
||||
timeout = "${var.lb_timeout_seconds}s"
|
||||
|
||||
vpc_access {
|
||||
connector = google_vpc_access_connector.this[0].id
|
||||
|
|
|
|||
|
|
@ -64,6 +64,7 @@ resource "google_compute_backend_service" "gateway" {
|
|||
name = "${local.name}-gateway-bs"
|
||||
protocol = "HTTP"
|
||||
load_balancing_scheme = "EXTERNAL_MANAGED"
|
||||
timeout_sec = var.lb_timeout_seconds
|
||||
|
||||
backend {
|
||||
group = google_compute_region_network_endpoint_group.gateway[0].id
|
||||
|
|
|
|||
|
|
@ -726,3 +726,14 @@ variable "collector_drain_timeout_seconds" {
|
|||
error_message = "collector_drain_timeout_seconds must be > 0."
|
||||
}
|
||||
}
|
||||
|
||||
variable "lb_timeout_seconds" {
|
||||
description = "Request timeout in seconds for the external load balancer backend service and the gateway Cloud Run service. Streams that stay silent longer than this (slow first token) are cut with a 504, so keep it at or above the proxy's request_timeout."
|
||||
type = number
|
||||
default = 600
|
||||
|
||||
validation {
|
||||
condition = var.lb_timeout_seconds >= 1
|
||||
error_message = "lb_timeout_seconds must be >= 1."
|
||||
}
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue