mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
* feat(infra): scale gateway on per-pod RPM and TPM in Helm and Terraform Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(helm): require the metrics server before rendering the gateway ServiceMonitor The http port serves /metrics/ behind virtual-key auth, so a ServiceMonitor pointed at it only collects 401s and the RPM/TPM HPA metrics never appear Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(infra): express gateway HPA, KEDA and ECS workload targets per second Rename the per-pod request and token targets in both Helm charts and the AWS module from per minute to per second, and shorten the recommended Prometheus rate window to [1m] with no * 60 so the adapter and KEDA signals are what the HPA compares against. ECS keeps CloudWatch's 60-second aggregation: the ALB target is 60x the per-second variable and the token metric math divides the period Sum by 60 before dividing by the running task count. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
106 lines
3.8 KiB
YAML
106 lines
3.8 KiB
YAML
suite: "keda"
|
|
templates:
|
|
- keda.yaml
|
|
release:
|
|
name: rel
|
|
namespace: llm
|
|
tests:
|
|
- it: "renders no scaled object by default"
|
|
asserts:
|
|
- hasDocuments: { count: 0 }
|
|
|
|
- it: "passes user triggers through and adds no prometheus triggers by default"
|
|
set:
|
|
keda.enabled: true
|
|
keda.triggers:
|
|
- type: cpu
|
|
metricType: Utilization
|
|
metadata: { value: "60" }
|
|
asserts:
|
|
- isKind: { of: ScaledObject }
|
|
- equal:
|
|
path: spec.triggers
|
|
value:
|
|
- type: cpu
|
|
metricType: Utilization
|
|
metadata: { value: "60" }
|
|
|
|
- it: "scales on release-wide requests per second divided by the per-replica target"
|
|
set:
|
|
keda.enabled: true
|
|
keda.prometheus.serverAddress: http://prometheus-operated.monitoring.svc:9090
|
|
keda.prometheus.requestsPerSecond: 90
|
|
asserts:
|
|
- lengthEqual: { path: spec.triggers, count: 1 }
|
|
- equal:
|
|
path: "spec.triggers[0]"
|
|
value:
|
|
type: prometheus
|
|
metadata:
|
|
serverAddress: http://prometheus-operated.monitoring.svc:9090
|
|
threshold: "90"
|
|
query: sum(rate(litellm_proxy_total_requests_metric_total{namespace="llm",job="rel-litellm"}[1m]))
|
|
|
|
- it: "scales on tokens per second on its own"
|
|
set:
|
|
keda.enabled: true
|
|
keda.prometheus.serverAddress: http://prom:9090
|
|
keda.prometheus.tokensPerSecond: 6000000
|
|
asserts:
|
|
- lengthEqual: { path: spec.triggers, count: 1 }
|
|
- equal: { path: "spec.triggers[0].type", value: prometheus }
|
|
- equal: { path: "spec.triggers[0].metadata.threshold", value: "6000000" }
|
|
- equal:
|
|
path: "spec.triggers[0].metadata.query"
|
|
value: sum(rate(litellm_total_tokens_metric_total{namespace="llm",job="rel-litellm"}[1m]))
|
|
|
|
- it: "appends requests and tokens triggers after user triggers and selects the metrics service job"
|
|
set:
|
|
keda.enabled: true
|
|
metricsServer.enabled: true
|
|
keda.triggers:
|
|
- type: cpu
|
|
metricType: Utilization
|
|
metadata: { value: "60" }
|
|
keda.prometheus.serverAddress: http://prom:9090
|
|
keda.prometheus.requestsPerSecond: 90
|
|
keda.prometheus.tokensPerSecond: 6000000
|
|
asserts:
|
|
- lengthEqual: { path: spec.triggers, count: 3 }
|
|
- equal: { path: "spec.triggers[0].type", value: cpu }
|
|
- equal: { path: "spec.triggers[1].metadata.threshold", value: "90" }
|
|
- equal:
|
|
path: "spec.triggers[1].metadata.query"
|
|
value: sum(rate(litellm_proxy_total_requests_metric_total{namespace="llm",job="rel-litellm-metrics"}[1m]))
|
|
- equal: { path: "spec.triggers[2].metadata.threshold", value: "6000000" }
|
|
- equal:
|
|
path: "spec.triggers[2].metadata.query"
|
|
value: sum(rate(litellm_total_tokens_metric_total{namespace="llm",job="rel-litellm-metrics"}[1m]))
|
|
- notMatchRegexRaw: { pattern: "\\* *60|per_minute|PerMinute" }
|
|
|
|
- it: "ignores the per-minute keys, which the chart never shipped"
|
|
set:
|
|
keda.enabled: true
|
|
keda.prometheus.serverAddress: http://prom:9090
|
|
keda.prometheus.requestsPerMinute: 5400
|
|
keda.prometheus.tokensPerMinute: 360000000
|
|
asserts:
|
|
- isKind: { of: ScaledObject }
|
|
- isNullOrEmpty: { path: spec.triggers }
|
|
|
|
- it: "refuses a workload target without a prometheus server address"
|
|
set:
|
|
keda.enabled: true
|
|
keda.prometheus.requestsPerSecond: 90
|
|
asserts:
|
|
- failedTemplate:
|
|
errorMessage: keda.prometheus.serverAddress is required when keda.prometheus.requestsPerSecond or tokensPerSecond is set
|
|
|
|
- it: "yields to the hpa when both autoscalers are enabled"
|
|
set:
|
|
autoscaling.enabled: true
|
|
keda.enabled: true
|
|
keda.prometheus.serverAddress: http://prom:9090
|
|
keda.prometheus.requestsPerSecond: 90
|
|
asserts:
|
|
- hasDocuments: { count: 0 }
|