mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
* feat(infra): scale gateway on per-pod RPM and TPM in Helm and Terraform Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(helm): require the metrics server before rendering the gateway ServiceMonitor The http port serves /metrics/ behind virtual-key auth, so a ServiceMonitor pointed at it only collects 401s and the RPM/TPM HPA metrics never appear Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(infra): express gateway HPA, KEDA and ECS workload targets per second Rename the per-pod request and token targets in both Helm charts and the AWS module from per minute to per second, and shorten the recommended Prometheus rate window to [1m] with no * 60 so the adapter and KEDA signals are what the HPA compares against. ECS keeps CloudWatch's 60-second aggregation: the ALB target is 60x the per-second variable and the token metric math divides the period Sum by 60 before dividing by the running task count. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
144 lines
5.3 KiB
YAML
144 lines
5.3 KiB
YAML
suite: "hpa"
|
|
templates:
|
|
- hpa.yaml
|
|
tests:
|
|
- it: "renders behavior when set"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.behavior:
|
|
scaleUp:
|
|
stabilizationWindowSeconds: 60
|
|
policies:
|
|
- type: Pods
|
|
value: 2
|
|
periodSeconds: 60
|
|
scaleDown:
|
|
stabilizationWindowSeconds: 90
|
|
policies:
|
|
- type: Pods
|
|
value: 1
|
|
periodSeconds: 60
|
|
asserts:
|
|
- isKind: { of: HorizontalPodAutoscaler }
|
|
- equal: { path: spec.behavior.scaleUp.stabilizationWindowSeconds, value: 60 }
|
|
- equal: { path: spec.behavior.scaleDown.stabilizationWindowSeconds, value: 90 }
|
|
|
|
- it: "does not render behavior when not set"
|
|
set:
|
|
autoscaling.enabled: true
|
|
asserts:
|
|
- isKind: { of: HorizontalPodAutoscaler }
|
|
- isNull: { path: spec.behavior }
|
|
|
|
- it: "scales on cpu at the documented 60 percent by default"
|
|
set:
|
|
autoscaling.enabled: true
|
|
asserts:
|
|
- isKind: { of: HorizontalPodAutoscaler }
|
|
- equal: { path: "spec.metrics[0].resource.name", value: cpu }
|
|
- equal: { path: "spec.metrics[0].resource.target.type", value: Utilization }
|
|
- equal: { path: "spec.metrics[0].resource.target.averageUtilization", value: 60 }
|
|
|
|
- it: "does not scale on memory by default"
|
|
set:
|
|
autoscaling.enabled: true
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 1 }
|
|
|
|
- it: "honours an explicit cpu target override"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetCPUUtilizationPercentage: 75
|
|
asserts:
|
|
- equal: { path: "spec.metrics[0].resource.target.averageUtilization", value: 75 }
|
|
|
|
- it: "renders a memory metric only when a memory target is set"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetMemoryUtilizationPercentage: 80
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 2 }
|
|
- equal: { path: "spec.metrics[1].resource.name", value: memory }
|
|
- equal: { path: "spec.metrics[1].resource.target.averageUtilization", value: 80 }
|
|
|
|
- it: "renders no workload metrics by default"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetMemoryUtilizationPercentage: 80
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 2 }
|
|
- notContains: { path: spec.metrics, content: { type: Pods }, any: true }
|
|
|
|
- it: "adds a requests-per-second Pods metric after the cpu metric"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetRequestsPerSecond: 90
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 2 }
|
|
- equal: { path: "spec.metrics[0].resource.name", value: cpu }
|
|
- equal:
|
|
path: "spec.metrics[1]"
|
|
value:
|
|
type: Pods
|
|
pods:
|
|
metric: { name: litellm_requests_per_second }
|
|
target: { type: AverageValue, averageValue: "90" }
|
|
|
|
- it: "adds a tokens-per-second Pods metric on its own"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetTokensPerSecond: 6M
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 2 }
|
|
- equal:
|
|
path: "spec.metrics[1]"
|
|
value:
|
|
type: Pods
|
|
pods:
|
|
metric: { name: litellm_tokens_per_second }
|
|
target: { type: AverageValue, averageValue: "6M" }
|
|
- notContains:
|
|
path: spec.metrics
|
|
content: { type: Pods, pods: { metric: { name: litellm_requests_per_second } } }
|
|
any: true
|
|
|
|
- it: "renders requests, tokens, cpu and memory metrics together"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetMemoryUtilizationPercentage: 80
|
|
autoscaling.targetRequestsPerSecond: 90
|
|
autoscaling.targetTokensPerSecond: 6000000
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 4 }
|
|
- equal: { path: "spec.metrics[0].resource.name", value: cpu }
|
|
- equal: { path: "spec.metrics[1].resource.name", value: memory }
|
|
- equal: { path: "spec.metrics[2].pods.metric.name", value: litellm_requests_per_second }
|
|
- equal: { path: "spec.metrics[2].pods.target.averageValue", value: "90" }
|
|
- equal: { path: "spec.metrics[3].pods.metric.name", value: litellm_tokens_per_second }
|
|
- equal: { path: "spec.metrics[3].pods.target.averageValue", value: "6000000" }
|
|
|
|
- it: "scales on workload metrics alone when the cpu target is cleared"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetCPUUtilizationPercentage: null
|
|
autoscaling.targetRequestsPerSecond: 90
|
|
autoscaling.targetTokensPerSecond: 6000000
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 2 }
|
|
- notContains: { path: spec.metrics, content: { type: Resource }, any: true }
|
|
- equal: { path: "spec.metrics[0].pods.metric.name", value: litellm_requests_per_second }
|
|
- equal: { path: "spec.metrics[1].pods.metric.name", value: litellm_tokens_per_second }
|
|
- notMatchRegexRaw: { pattern: per_minute }
|
|
|
|
- it: "ignores the per-minute keys, which the chart never shipped"
|
|
set:
|
|
autoscaling.enabled: true
|
|
autoscaling.targetRequestsPerMinute: 5400
|
|
autoscaling.targetTokensPerMinute: 360000000
|
|
asserts:
|
|
- lengthEqual: { path: spec.metrics, count: 1 }
|
|
- notContains: { path: spec.metrics, content: { type: Pods }, any: true }
|
|
|
|
- it: "renders no hpa when autoscaling is disabled"
|
|
asserts:
|
|
- hasDocuments: { count: 0 }
|