mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
* feat(infra): scale gateway on per-pod RPM and TPM in Helm and Terraform Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(helm): require the metrics server before rendering the gateway ServiceMonitor The http port serves /metrics/ behind virtual-key auth, so a ServiceMonitor pointed at it only collects 401s and the RPM/TPM HPA metrics never appear Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(infra): express gateway HPA, KEDA and ECS workload targets per second Rename the per-pod request and token targets in both Helm charts and the AWS module from per minute to per second, and shorten the recommended Prometheus rate window to [1m] with no * 60 so the adapter and KEDA signals are what the HPA compares against. ECS keeps CloudWatch's 60-second aggregation: the ALB target is 60x the per-second variable and the token metric math divides the period Sum by 60 before dividing by the running task count. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
58 lines
2.2 KiB
YAML
58 lines
2.2 KiB
YAML
{{- if and .Values.keda.enabled (not .Values.autoscaling.enabled) }}
|
|
apiVersion: keda.sh/v1alpha1
|
|
kind: ScaledObject
|
|
metadata:
|
|
name: {{ include "litellm.fullname" . }}
|
|
labels:
|
|
{{- include "litellm.labels" . | nindent 4 }}
|
|
{{- if .Values.keda.scaledObject.annotations }}
|
|
annotations: {{ toYaml .Values.keda.scaledObject.annotations | nindent 4 }}
|
|
{{- end }}
|
|
spec:
|
|
scaleTargetRef:
|
|
name: {{ include "litellm.fullname" . }}
|
|
pollingInterval: {{ .Values.keda.pollingInterval }}
|
|
cooldownPeriod: {{ .Values.keda.cooldownPeriod }}
|
|
minReplicaCount: {{ .Values.keda.minReplicas }}
|
|
maxReplicaCount: {{ .Values.keda.maxReplicas }}
|
|
{{- with .Values.keda.fallback }}
|
|
fallback:
|
|
failureThreshold: {{ .failureThreshold | default 3 }}
|
|
replicas: {{ .replicas | default $.Values.keda.maxReplicas }}
|
|
{{- end }}
|
|
triggers:
|
|
{{- with .Values.keda.triggers }}
|
|
{{- toYaml . | nindent 2 }}
|
|
{{- end }}
|
|
{{- $prom := .Values.keda.prometheus }}
|
|
{{- if or $prom.requestsPerSecond $prom.tokensPerSecond }}
|
|
{{- if not $prom.serverAddress }}
|
|
{{- fail "keda.prometheus.serverAddress is required when keda.prometheus.requestsPerSecond or tokensPerSecond is set" }}
|
|
{{- end }}
|
|
{{- $selector := printf "namespace=%q,job=%q" .Release.Namespace (printf "%s%s" (include "litellm.fullname" .) (ternary "-metrics" "" .Values.metricsServer.enabled)) }}
|
|
{{- with $prom.requestsPerSecond }}
|
|
- type: prometheus
|
|
metadata:
|
|
serverAddress: {{ $prom.serverAddress | quote }}
|
|
threshold: {{ toJson . | trimAll "\"" | quote }}
|
|
query: {{ printf "sum(rate(litellm_proxy_total_requests_metric_total{%s}[1m]))" $selector | quote }}
|
|
{{- end }}
|
|
{{- with $prom.tokensPerSecond }}
|
|
- type: prometheus
|
|
metadata:
|
|
serverAddress: {{ $prom.serverAddress | quote }}
|
|
threshold: {{ toJson . | trimAll "\"" | quote }}
|
|
query: {{ printf "sum(rate(litellm_total_tokens_metric_total{%s}[1m]))" $selector | quote }}
|
|
{{- end }}
|
|
{{- end }}
|
|
advanced:
|
|
restoreToOriginalReplicaCount: {{ .Values.keda.restoreToOriginalReplicaCount }}
|
|
{{- if .Values.keda.behavior }}
|
|
horizontalPodAutoscalerConfig:
|
|
behavior:
|
|
{{- with .Values.keda.behavior }}
|
|
{{- toYaml . | nindent 8 }}
|
|
{{- end }}
|
|
|
|
{{- end }}
|
|
{{- end }}
|