diff --git a/.circleci/config.yml b/.circleci/config.yml index 9b1e433be93..973d6b49e01 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -1704,30 +1704,36 @@ jobs: IMAGE_TAG=${CIRCLE_SHA1:-ci} kind load docker-image litellm-ci:${IMAGE_TAG} --name litellm-test - # Run helm lint - run: name: Run helm lint command: | - helm lint ./helm/litellm-helm + helm lint ./helm/litellm -f ./helm/litellm/ci/test-values.yaml + + - run: + name: Start PostgreSQL for the chart + command: | + kubectl apply -f ./helm/litellm/ci/postgres.yaml + kubectl rollout status deployment/litellm-ci-postgres --timeout=3m - # Run helm tests - run: name: Run helm tests command: | IMAGE_TAG=${CIRCLE_SHA1:-ci} - helm install litellm ./helm/litellm-helm -f ./helm/litellm-helm/ci/test-values.yaml \ + helm install litellm ./helm/litellm -f ./helm/litellm/ci/test-values.yaml \ --set image.repository=litellm-ci \ --set image.tag=${IMAGE_TAG} \ - --set image.pullPolicy=Never - # Wait for pod to be ready - echo "Waiting 30 seconds for pod to be ready..." - sleep 30 + --set image.pullPolicy=Never \ + --wait --wait-for-jobs --timeout 10m || { + kubectl get pods -o wide + kubectl describe pods -l app.kubernetes.io/instance=litellm + kubectl logs -l app.kubernetes.io/component=proxy --tail=200 || true + kubectl logs -l app.kubernetes.io/component=migrations --tail=200 || true + exit 1 + } - # Print pod logs before running tests - echo "Printing pod logs..." - kubectl logs $(kubectl get pods -l app.kubernetes.io/name=litellm -o jsonpath="{.items[0].metadata.name}") + echo "Printing proxy pod logs..." + kubectl logs -l app.kubernetes.io/component=proxy --tail=200 - # Run the helm tests helm test litellm --logs # Cleanup diff --git a/.github/workflows/helm_unit_test.yml b/.github/workflows/helm_unit_test.yml index f95848945a0..f6fcd0f7bb7 100644 --- a/.github/workflows/helm_unit_test.yml +++ b/.github/workflows/helm_unit_test.yml @@ -41,14 +41,13 @@ jobs: - name: Run unit tests run: | - for chart in helm/litellm-helm helm/litellm; do - declared="$(grep -h '^suite:' "$chart"/tests/*.yaml | wc -l | tr -d '[:space:]')" - output="$(mktemp)" - helm unittest -f 'tests/*.yaml' "$chart" | tee "$output" - executed="$(sed -n 's/^Test Suites:.*[[:space:]]\([0-9][0-9]*\) total$/\1/p' "$output")" - if [ "$declared" != "$executed" ]; then - echo "::error::$chart declares $declared test suites but helm-unittest ran $executed. Suites are being skipped silently, so their assertions never execute." - exit 1 - fi - echo "$chart: all $declared declared test suites ran" - done + chart=helm/litellm + declared="$(grep -h '^suite:' "$chart"/tests/*.yaml | wc -l | tr -d '[:space:]')" + output="$(mktemp)" + helm unittest -f 'tests/*.yaml' "$chart" | tee "$output" + executed="$(sed -n 's/^Test Suites:.*[[:space:]]\([0-9][0-9]*\) total$/\1/p' "$output")" + if [ "$declared" != "$executed" ]; then + echo "::error::$chart declares $declared test suites but helm-unittest ran $executed. Suites are being skipped silently, so their assertions never execute." + exit 1 + fi + echo "$chart: all $declared declared test suites ran" diff --git a/.gitignore b/.gitignore index 7da917ce450..cd74b5b7e1f 100644 --- a/.gitignore +++ b/.gitignore @@ -60,7 +60,7 @@ ui/litellm-dashboard/node_modules ui/litellm-dashboard/next-env.d.ts ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json -helm/litellm-helm/*.tgz +helm/litellm/charts/*.tgz helm/*.tgz litellm/proxy/vertex_key.json **/.vim/ diff --git a/Makefile b/Makefile index cad3242fbce..8e4e0344004 100644 --- a/Makefile +++ b/Makefile @@ -352,7 +352,7 @@ test-integration: install-test-deps $(UV_RUN) pytest tests/ -k "not test_litellm" test-unit-helm: install-helm-unittest - helm unittest -f 'tests/*.yaml' helm/litellm-helm + helm unittest -f 'tests/*.yaml' helm/litellm # LLM Translation testing targets test-llm-translation: install-test-deps diff --git a/helm/litellm-helm/.helmignore b/helm/litellm-helm/.helmignore deleted file mode 100644 index 0e8a0eb36f4..00000000000 --- a/helm/litellm-helm/.helmignore +++ /dev/null @@ -1,23 +0,0 @@ -# Patterns to ignore when building packages. -# This supports shell glob matching, relative path matching, and -# negation (prefixed with !). Only one pattern per line. -.DS_Store -# Common VCS dirs -.git/ -.gitignore -.bzr/ -.bzrignore -.hg/ -.hgignore -.svn/ -# Common backup files -*.swp -*.bak -*.tmp -*.orig -*~ -# Various IDEs -.project -.idea/ -*.tmproj -.vscode/ diff --git a/helm/litellm-helm/Chart.lock b/helm/litellm-helm/Chart.lock deleted file mode 100644 index d626fbb472b..00000000000 --- a/helm/litellm-helm/Chart.lock +++ /dev/null @@ -1,9 +0,0 @@ -dependencies: -- name: postgresql - repository: oci://registry-1.docker.io/bitnamicharts - version: 14.3.1 -- name: redis - repository: oci://registry-1.docker.io/bitnamicharts - version: 18.19.1 -digest: sha256:38962e231f6596b93f82a8412bbe4cf5de696caecf5775dfbbd163383eb1c009 -generated: "2026-07-28T10:21:22.511401-07:00" diff --git a/helm/litellm-helm/Chart.yaml b/helm/litellm-helm/Chart.yaml deleted file mode 100644 index a3cb388ffc6..00000000000 --- a/helm/litellm-helm/Chart.yaml +++ /dev/null @@ -1,41 +0,0 @@ -apiVersion: v2 - -# We can't call ourselves just "litellm" because then we couldn't publish to the -# same OCI repository as the "litellm" OCI image -name: litellm-helm -description: Call all LLM APIs using the OpenAI format - -# A chart can be either an 'application' or a 'library' chart. -# -# Application charts are a collection of templates that can be packaged into versioned archives -# to be deployed. -# -# Library charts provide useful utilities or functions for the chart developer. They're included as -# a dependency of application charts to inject those utilities and functions into the rendering -# pipeline. Library charts do not define any templates and therefore cannot be deployed. -type: application - -# This is the chart version. This version number should be incremented each time you make changes -# to the chart and its templates, including the app version. -# Versions are expected to follow Semantic Versioning (https://semver.org/) -version: 1.1.3 - -# This is the version number of the application being deployed. This version number should be -# incremented each time you make changes to the application. Versions are not expected to -# follow Semantic Versioning. They should reflect the version the application is using. -# It is recommended to use it with quotes. -appVersion: v1.85.1 - -annotations: - org.opencontainers.image.source: "https://github.com/BerriAI/litellm" - org.opencontainers.image.url: "https://docs.litellm.ai/" - -dependencies: - - name: "postgresql" - version: "14.3.1" - repository: oci://registry-1.docker.io/bitnamicharts - condition: db.deployStandalone - - name: redis - version: "18.19.1" - repository: oci://registry-1.docker.io/bitnamicharts - condition: redis.enabled diff --git a/helm/litellm-helm/README.md b/helm/litellm-helm/README.md deleted file mode 100644 index bf4089404db..00000000000 --- a/helm/litellm-helm/README.md +++ /dev/null @@ -1,227 +0,0 @@ -# Helm Chart for LiteLLM - -> [!IMPORTANT] -> This is community maintained, Please make an issue if you run into a bug -> We recommend using [Docker or Kubernetes for production deployments](https://docs.litellm.ai/docs/proxy/prod) - -## Prerequisites - -- Kubernetes 1.21+ -- Helm 3.8.0+ - -If `db.deployStandalone` is used: - -- PV provisioner support in the underlying infrastructure - -If `db.useStackgresOperator` is used (not yet implemented): - -- The Stackgres Operator must already be installed in the Kubernetes Cluster. This chart will **not** install the operator if it is missing. - -## Parameters - -### LiteLLM Proxy Deployment Settings - -| Name | Description | Value | -| --------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------- | -| `replicaCount` | The number of LiteLLM Proxy pods to be deployed | `1` | -| `masterkeySecretName` | The name of the Kubernetes Secret that contains the Master API Key for LiteLLM. If not specified, use the generated secret name. | N/A | -| `masterkeySecretKey` | The key within the Kubernetes Secret that contains the Master API Key for LiteLLM. If not specified, use `masterkey` as the key. | N/A | -| `masterkey` | The Master API Key for LiteLLM. If not specified, a random key in the `sk-...` format is generated on first install and reused on upgrades. | N/A | -| `environmentSecrets` | An optional array of Secret object names. The keys and values in these secrets will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` | -| `environmentConfigMaps` | An optional array of ConfigMap object names. The keys and values in these configmaps will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` | -| `image.repository` | LiteLLM Proxy image repository | `ghcr.io/berriai/litellm` | -| `image.pullPolicy` | LiteLLM Proxy image pull policy | `IfNotPresent` | -| `image.tag` | Overrides the image tag whose default the latest version of LiteLLM at the time this chart was published. | `""` | -| `imagePullSecrets` | Registry credentials for the LiteLLM and initContainer images. | `[]` | -| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` | -| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` | -| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` | -| `livenessProbe.*` | Liveness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | -| `readinessProbe.*` | Readiness probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | -| `startupProbe.*` | Startup probe settings for the LiteLLM container (`path`, `periodSeconds`, `timeoutSeconds`, thresholds, and initial delay). | See `values.yaml` | -| `resources.*` | CPU/memory requests and limits for the LiteLLM container. Unset by default; production deployments should set 1 CPU and 4Gi of memory per worker. | `{}` | -| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` | -| `ingress.labels` | Additional labels for the Ingress resource | `{}` | -| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A | -| `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` | -| `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` | -| `proxyConfigMap.key` | Key in the ConfigMap that contains the proxy config file. | `"config.yaml"` | -| `proxy_config.*` | See [values.yaml](./values.yaml) for default settings. Rendered into the ConfigMap’s `config.yaml` only when `proxyConfigMap.create=true`. See [example_config_yaml](../../../litellm/proxy/example_config_yaml/) for configuration examples. | `N/A` | -| `extraContainers[]` | An array of additional containers to be deployed as sidecars alongside the LiteLLM Proxy. | -| `pdb.enabled` | Enable a PodDisruptionBudget for the LiteLLM proxy Deployment | `false` | -| `pdb.minAvailable` | Minimum number/percentage of pods that must be available during **voluntary** disruptions (choose **one** of minAvailable/maxUnavailable) | `null` | -| `pdb.maxUnavailable` | Maximum number/percentage of pods that can be unavailable during **voluntary** disruptions (choose **one** of minAvailable/maxUnavailable) | `null` | -| `pdb.annotations` | Extra metadata annotations to add to the PDB | `{}` | -| `pdb.labels` | Extra metadata labels to add to the PDB | `{}` | - -| `billingMetrics.enabled` | Enable enterprise billable-request metering. Requires an enterprise license. | `false` | -| `billingMetrics.endpoint` | Collector that the billable-request counter is pushed to. | `https://telemetry.litellm.ai` | -| `billingMetrics.secretName` | Name of an existing Secret holding the mTLS client certificate, under the keys `tls.crt` and `tls.key`. | `litellm-billing-metrics-mtls` | -| `billingMetrics.caSecretName` | Name of an existing Secret holding a CA bundle under the key `ca.crt`. Only needed for a private or test collector whose server certificate is not on the public web PKI. | `""` | -| `billingMetrics.exportIntervalMs` | How often the counter is pushed, in milliseconds. The proxy defaults to `60000` when unset. | `""` | - -#### Example `proxy_config` ConfigMap from values (default): - -``` -proxyConfigMap: - create: true - key: "config.yaml" - -proxy_config: - general_settings: - master_key: os.environ/PROXY_MASTER_KEY - model_list: - - model_name: gpt-3.5-turbo - litellm_params: - model: gpt-3.5-turbo - api_key: eXaMpLeOnLy -``` - -#### Example using existing `proxyConfigMap` instead of creating it: - -``` -proxyConfigMap: - create: false - name: my-litellm-config - key: config.yaml - -# proxy_config is ignored in this mode -``` - -#### Example `environmentSecrets` Secret - -``` -apiVersion: v1 -kind: Secret -metadata: - name: litellm-envsecrets -data: - AZURE_OPENAI_API_KEY: TXlTZWN1cmVLM3k= -type: Opaque -``` - -#### Enterprise billable-request metering - -Enterprise licenses meter billable requests by pushing a counter to LiteLLM's collector over mutual TLS. The chart does not create the client certificate; it mounts one you already hold, read-only, so the private key is never exposed through the environment. Create the Secret under the name the chart expects, then turn the block on: - -``` -kubectl create secret tls litellm-billing-metrics-mtls --cert=client.crt --key=client.key -``` - -``` -billingMetrics: - enabled: true -``` - -Set `billingMetrics.caSecretName` only when the collector is a private or test one whose server certificate is not on the public web PKI; the production collector needs no CA override. The chart fails the render rather than deploying a proxy that silently never exports, so a missing `secretName` or an emptied `endpoint` surfaces at `helm install` time. - -### Database Settings - -| Name | Description | Value | -| ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ | -| `db.useExisting` | Use an existing Postgres database. A Kubernetes Secret object must exist that contains credentials for connecting to the database. An example secret object definition is provided below. | `false` | -| `db.endpoint` | If `db.useExisting` is `true`, this is the IP, Hostname or Service Name of the Postgres server to connect to. | `localhost` | -| `db.database` | If `db.useExisting` is `true`, the name of the existing database to connect to. | `litellm` | -| `db.url` | If `db.useExisting` is `true`, the connection url of the existing database to connect to can be overwritten with this value. | `postgresql://$(DATABASE_USERNAME):$(DATABASE_PASSWORD)@$(DATABASE_HOST)/$(DATABASE_NAME)` | -| `db.secret.name` | If `db.useExisting` is `true`, the name of the Kubernetes Secret that contains credentials. | `postgres` | -| `db.secret.usernameKey` | If `db.useExisting` is `true`, the name of the key within the Kubernetes Secret that holds the username for authenticating with the Postgres instance. | `username` | -| `db.secret.passwordKey` | If `db.useExisting` is `true`, the name of the key within the Kubernetes Secret that holds the password associates with the above user. | `password` | -| `db.useStackgresOperator` | Not yet implemented. | `false` | -| `db.deployStandalone` | Deploy a standalone, single instance deployment of Postgres, using the Bitnami postgresql chart. This is useful for getting started but doesn't provide HA or (by default) data backups. | `true` | -| `postgresql.*` | If `db.deployStandalone` is `true`, configuration passed to the Bitnami postgresql chart. See the [Bitnami Documentation](https://github.com/bitnami/charts/tree/main/bitnami/postgresql) for full configuration details. See [values.yaml](./values.yaml) for the default configuration. | See [values.yaml](./values.yaml) | -| `postgresql.auth.*` | If `db.deployStandalone` is `true`, care should be taken to ensure the default `password` and `postgres-password` values are **NOT** used. | `NoTaGrEaTpAsSwOrD` | -| `postgresql.image.*` | If `db.deployStandalone` is `true`, the image for the bundled Postgres. Pinned to a `docker.io/bitnamilegacy` build because Bitnami retired the versioned tags under `docker.io/bitnami`. | `bitnamilegacy/postgresql:16.2.0-debian-12-r6` | -| `redis.image.*` | If `redis.enabled` is `true`, the image for the bundled Redis. Pinned to a `docker.io/bitnamilegacy` build for the same reason. | `bitnamilegacy/redis:7.2.4-debian-12-r9` | - -#### Bundled Postgres image - -Bitnami removed the versioned tags from `docker.io/bitnami` and republished the archived builds under `docker.io/bitnamilegacy`, so the image defaults that ship inside the `postgresql` and `redis` subcharts no longer pull. The chart pins both to the `bitnamilegacy` copies of the exact builds those subchart versions were released with, which keeps the on-disk data directory layout unchanged for existing installs. - -Keep `postgresql.image.tag` pinned. `docker.io/bitnami/postgresql` still publishes a floating `latest`, and pointing the bundled Postgres at a different major version starts the server against a data directory it cannot read (`database files are incompatible with server`). There is no in-place way back, so crossing a major version means dumping the database with the old image and restoring it into the new one. The chart refuses to render when the tag is empty or `latest`. - -Those images no longer receive updates. For anything beyond getting started, run Postgres outside the chart and point at it with `db.useExisting`. - -#### Example Postgres `db.useExisting` Secret - -```yaml -apiVersion: v1 -kind: Secret -metadata: - name: postgres -data: - # Password for the "postgres" user - postgres-password: - username: litellm - password: -type: Opaque -``` - -#### Examples for `environmentSecrets` and `environemntConfigMaps` - -```yaml -# Use config map for not-secret configuration data -apiVersion: v1 -kind: ConfigMap -metadata: - name: litellm-env-configmap -data: - SOME_KEY: someValue - ANOTHER_KEY: anotherValue -``` - -```yaml -# Use secrets for things which are actually secret like API keys, credentials, etc -# Base64 encode the values stored in a Kubernetes Secret: $ pbpaste | base64 | pbcopy -# The --decode flag is convenient: $ pbpaste | base64 --decode - -apiVersion: v1 -kind: Secret -metadata: - name: litellm-env-secret -type: Opaque -data: - SOME_PASSWORD: cDZbUGVXeU5e0ZW # base64 encoded - ANOTHER_PASSWORD: AAZbUGVXeU5e0ZB # base64 encoded -``` - -Source: [GitHub Gist from troyharvey](https://gist.github.com/troyharvey/4506472732157221e04c6b15e3b3f094) - -### Migration Job Settings - -The migration job supports both ArgoCD and Helm hooks to ensure database migrations run at the appropriate time during deployments. - -| Name | Description | Value | -| -------------------------------------- | -------------------------------------------------------------------------------------------------------------------- | ------- | -| `migrationJob.enabled` | Enable or disable the schema migration Job | `true` | -| `migrationJob.backoffLimit` | Backoff limit for Job restarts | `4` | -| `migrationJob.ttlSecondsAfterFinished` | TTL for completed migration jobs | `120` | -| `migrationJob.annotations` | Additional annotations for the migration job pod | `{}` | -| `migrationJob.extraContainers` | Additional containers to run alongside the migration job | `[]` | -| `migrationJob.hooks.argocd.enabled` | Enable ArgoCD hooks for the migration job (uses PreSync hook with BeforeHookCreation delete policy) | `true` | -| `migrationJob.hooks.helm.enabled` | Enable Helm hooks for the migration job (uses pre-install,pre-upgrade hooks with before-hook-creation delete policy) | `false` | -| `migrationJob.hooks.helm.weight` | Helm hook execution order (lower weights executed first). Optional - defaults to "1" if not specified. | N/A | - -## Accessing the Admin UI - -When browsing to the URL published per the settings in `ingress.*`, you will -be prompted for **Admin Configuration**. The **Proxy Endpoint** is the internal -(from the `litellm` pod's perspective) URL published by the `-litellm` -Kubernetes Service. If the deployment uses the default settings for this -service, the **Proxy Endpoint** should be set to `http://-litellm:4000`. - -The **Proxy Key** is the value specified for `masterkey` or, if a `masterkey` -was not provided to the helm command line, the `masterkey` is a randomly -generated string in the `sk-...` format stored in the `-litellm-masterkey` Kubernetes Secret. -The key is generated once on the first install; later `helm upgrade` runs reuse the -value already in that Secret, so upgrading never rotates the master key. - -```bash -kubectl -n litellm get secret -litellm-masterkey -o jsonpath="{.data.masterkey}" -``` - -## Admin UI Limitations - -At the time of writing, the Admin UI is unable to add models. This is because -it would need to update the `config.yaml` file which is a exposed ConfigMap, and -therefore, read-only. This is a limitation of this helm chart, not the Admin UI -itself. diff --git a/helm/litellm-helm/charts/postgresql-14.3.1.tgz b/helm/litellm-helm/charts/postgresql-14.3.1.tgz deleted file mode 100644 index e8e2fac0fda..00000000000 Binary files a/helm/litellm-helm/charts/postgresql-14.3.1.tgz and /dev/null differ diff --git a/helm/litellm-helm/charts/redis-18.19.1.tgz b/helm/litellm-helm/charts/redis-18.19.1.tgz deleted file mode 100644 index 4a55a980088..00000000000 Binary files a/helm/litellm-helm/charts/redis-18.19.1.tgz and /dev/null differ diff --git a/helm/litellm-helm/ci/test-values.yaml b/helm/litellm-helm/ci/test-values.yaml deleted file mode 100644 index 33a4df942ee..00000000000 --- a/helm/litellm-helm/ci/test-values.yaml +++ /dev/null @@ -1,15 +0,0 @@ -fullnameOverride: "" -# Disable database deployment and configuration -db: - deployStandalone: false - useExisting: false - -# Test environment variables -envVars: - DD_ENV: "dev_helm" - DD_SERVICE: "litellm" - USE_DDTRACE: "true" - -# Disable migration job since we're not using a database -migrationJob: - enabled: false \ No newline at end of file diff --git a/helm/litellm-helm/templates/NOTES.txt b/helm/litellm-helm/templates/NOTES.txt deleted file mode 100644 index 017bbfa78bd..00000000000 --- a/helm/litellm-helm/templates/NOTES.txt +++ /dev/null @@ -1,23 +0,0 @@ -1. Get the application URL by running these commands: -{{- if .Values.ingress.enabled }} -{{- range $host := .Values.ingress.hosts }} - {{- range .paths }} - http{{ if $.Values.ingress.tls }}s{{ end }}://{{ $host.host }}{{ .path }} - {{- end }} -{{- end }} -{{- else if contains "NodePort" .Values.service.type }} - export NODE_PORT=$(kubectl get --namespace {{ .Release.Namespace }} -o jsonpath="{.spec.ports[0].nodePort}" services {{ include "litellm.fullname" . }}) - export NODE_IP=$(kubectl get nodes --namespace {{ .Release.Namespace }} -o jsonpath="{.items[0].status.addresses[0].address}") - echo http://$NODE_IP:$NODE_PORT -{{- else if contains "LoadBalancer" .Values.service.type }} - NOTE: It may take a few minutes for the LoadBalancer IP to be available. - You can watch the status of by running 'kubectl get --namespace {{ .Release.Namespace }} svc -w {{ include "litellm.fullname" . }}' - export SERVICE_IP=$(kubectl get svc --namespace {{ .Release.Namespace }} {{ include "litellm.fullname" . }} --template "{{"{{ range (index .status.loadBalancer.ingress 0) }}{{.}}{{ end }}"}}") - echo http://$SERVICE_IP:{{ .Values.service.port }} -{{- else if contains "ClusterIP" .Values.service.type }} - export POD_NAME=$(kubectl get pods --namespace {{ .Release.Namespace }} -l "app.kubernetes.io/name={{ include "litellm.name" . }},app.kubernetes.io/instance={{ .Release.Name }}" -o jsonpath="{.items[0].metadata.name}") - export CONTAINER_PORT=$(kubectl get pod --namespace {{ .Release.Namespace }} $POD_NAME -o jsonpath="{.spec.containers[0].ports[0].containerPort}") - echo "Visit http://127.0.0.1:8080 to use your application" - kubectl --namespace {{ .Release.Namespace }} port-forward $POD_NAME 8080:$CONTAINER_PORT -{{- end }} -PDB: {{ if .Values.pdb.enabled }}enabled{{ else }}disabled{{ end }}. Configure via .Values.pdb.* \ No newline at end of file diff --git a/helm/litellm-helm/templates/_helpers.tpl b/helm/litellm-helm/templates/_helpers.tpl deleted file mode 100644 index 9630633912e..00000000000 --- a/helm/litellm-helm/templates/_helpers.tpl +++ /dev/null @@ -1,323 +0,0 @@ -{{/* -Expand the name of the chart. -*/}} -{{- define "litellm.name" -}} -{{- default .Chart.Name .Values.nameOverride | trunc 63 | trimSuffix "-" }} -{{- end }} - -{{/* -Create a default fully qualified app name. -We truncate at 63 chars because some Kubernetes name fields are limited to this (by the DNS naming spec). -If release name contains chart name it will be used as a full name. -*/}} -{{- define "litellm.fullname" -}} -{{- if .Values.fullnameOverride }} -{{- .Values.fullnameOverride | trunc 63 | trimSuffix "-" }} -{{- else }} -{{- $name := default .Chart.Name .Values.nameOverride }} -{{- if contains $name .Release.Name }} -{{- .Release.Name | trunc 63 | trimSuffix "-" }} -{{- else }} -{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" }} -{{- end }} -{{- end }} -{{- end }} - -{{/* -Create chart name and version as used by the chart label. -*/}} -{{- define "litellm.chart" -}} -{{- printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} -{{- end }} - -{{/* -Common labels -*/}} -{{- define "litellm.labels" -}} -helm.sh/chart: {{ include "litellm.chart" . }} -{{ include "litellm.selectorLabels" . }} -{{- if .Chart.AppVersion }} -app.kubernetes.io/version: {{ .Chart.AppVersion | quote }} -{{- end }} -app.kubernetes.io/managed-by: {{ .Release.Service }} -{{- end }} - -{{/* -Selector labels -*/}} -{{- define "litellm.selectorLabels" -}} -app.kubernetes.io/name: {{ include "litellm.name" . }} -app.kubernetes.io/instance: {{ .Release.Name }} -{{- end }} - -{{/* -Enterprise billable-request metering. The client certificate identifies the -deployment to LiteLLM's collector, so it is mounted read-only from an existing -Secret rather than passed through the environment. -*/}} -{{- define "litellm.billingMetrics.certDir" -}}/etc/litellm/billing-mtls{{- end -}} -{{- define "litellm.billingMetrics.caDir" -}}/etc/litellm/billing-mtls-ca{{- end -}} - -{{- define "litellm.billingMetricsEnv" -}} -- name: LITELLM_BILLING_METRICS_ENDPOINT - value: {{ required "billingMetrics.endpoint is required when billingMetrics.enabled is true" .Values.billingMetrics.endpoint | quote }} -- name: LITELLM_BILLING_METRICS_CLIENT_CERT - value: {{ printf "%s/tls.crt" (include "litellm.billingMetrics.certDir" .) | quote }} -- name: LITELLM_BILLING_METRICS_CLIENT_KEY - value: {{ printf "%s/tls.key" (include "litellm.billingMetrics.certDir" .) | quote }} -{{- if .Values.billingMetrics.caSecretName }} -- name: LITELLM_BILLING_METRICS_CA_CERT - value: {{ printf "%s/ca.crt" (include "litellm.billingMetrics.caDir" .) | quote }} -{{- end }} -{{- with .Values.billingMetrics.exportIntervalMs }} -- name: LITELLM_BILLING_METRICS_EXPORT_INTERVAL_MS - value: {{ . | quote }} -{{- end }} -{{- end -}} - -{{- define "litellm.billingMetricsVolumes" -}} -- name: billing-metrics-mtls - secret: - secretName: {{ required "billingMetrics.secretName is required when billingMetrics.enabled is true (an existing Secret with tls.crt and tls.key)" .Values.billingMetrics.secretName }} -{{- if .Values.billingMetrics.caSecretName }} -- name: billing-metrics-mtls-ca - secret: - secretName: {{ .Values.billingMetrics.caSecretName }} -{{- end }} -{{- end -}} - -{{- define "litellm.billingMetricsVolumeMounts" -}} -- name: billing-metrics-mtls - mountPath: {{ include "litellm.billingMetrics.certDir" . }} - readOnly: true -{{- if .Values.billingMetrics.caSecretName }} -- name: billing-metrics-mtls-ca - mountPath: {{ include "litellm.billingMetrics.caDir" . }} - readOnly: true -{{- end }} -{{- end -}} - -{{/* -Create the name of the service account to use -*/}} -{{- define "litellm.serviceAccountName" -}} -{{- if .Values.serviceAccount.create }} -{{- default (include "litellm.fullname" .) .Values.serviceAccount.name }} -{{- else }} -{{- default "default" .Values.serviceAccount.name }} -{{- end }} -{{- end }} - -{{/* -Create the service account name used by migration jobs. -When Helm hooks are enabled, pre-install/pre-upgrade hooks run before normal resources. -If this chart is creating the ServiceAccount, it is not yet available for the hook job, -so fall back to "default" (or an explicit override) to avoid a cyclic dependency. -*/}} -{{- define "litellm.migrationServiceAccountName" -}} -{{- if and .Values.migrationJob.hooks.helm.enabled .Values.serviceAccount.create }} -{{- default "default" .Values.migrationJob.serviceAccountName }} -{{- else }} -{{- include "litellm.serviceAccountName" . }} -{{- end }} -{{- end }} - -{{/* -Get redis service name. -The bundled Redis subchart only serves sentinel in "replication" architecture -(it rejects standalone + sentinel outright), and in that mode the sentinel -Service is named "-redis", not "-redis-master". -*/}} -{{- define "litellm.redis.serviceName" -}} -{{- if .Values.redis.sentinel.enabled -}} -{{- printf "%s-%s" .Release.Name (default "redis" .Values.redis.nameOverride | trunc 63 | trimSuffix "-") -}} -{{- else -}} -{{- printf "%s-%s-master" .Release.Name (default "redis" .Values.redis.nameOverride | trunc 63 | trimSuffix "-") -}} -{{- end -}} -{{- end -}} - -{{/* -Get redis service port -*/}} -{{- define "litellm.redis.port" -}} -{{- if .Values.redis.sentinel.enabled -}} -{{ .Values.redis.sentinel.service.ports.sentinel }} -{{- else -}} -{{ .Values.redis.master.service.ports.redis }} -{{- end -}} -{{- end -}} - -{{/* -Reject an unpinned image tag for the bundled PostgreSQL. -A floating tag lets a chart upgrade start a newer PostgreSQL major against the -existing PersistentVolumeClaim. The server then refuses to start on a data -directory written by another major version, and the only way back is a dump -taken before the change, which by that point no longer exists. -*/}} -{{- define "litellm.validateBundledPostgresImageTag" -}} -{{- $tag := .Values.postgresql.image.tag | default "" | toString -}} -{{- $digest := .Values.postgresql.image.digest | default "" | toString -}} -{{- if and (eq $digest "") (or (eq $tag "") (eq $tag "latest")) -}} -{{- fail (printf "postgresql.image.tag must be pinned to an explicit version when db.deployStandalone is true (got %q). An unpinned tag can start a different PostgreSQL major against the existing data directory, which makes the database unreadable and is not recoverable in place. Crossing a major version requires a dump and restore." $tag) -}} -{{- end -}} -{{- end -}} - -{{/* -Environment shared by the proxy container and the opt-in collector sidecar: -database, pgbouncer, master key, redis, user envVars. Both containers must see -the same DATABASE_URL and REDIS_* so the sidecar reaches the pod's pgbouncer -and the same spend transaction buffer. -*/}} -{{- define "litellm.proxyEnv" -}} -- name: HOST - value: "{{ .Values.listen | default "0.0.0.0" }}" -- name: PORT - value: {{ .Values.service.port | quote}} -{{- if .Values.db.deployStandalone }} -- name: DATABASE_USERNAME - valueFrom: - secretKeyRef: - name: {{ include "litellm.fullname" . }}-dbcredentials - key: username -- name: DATABASE_PASSWORD - valueFrom: - secretKeyRef: - name: {{ include "litellm.fullname" . }}-dbcredentials - key: password -- name: DATABASE_HOST - value: {{ .Release.Name }}-postgresql -- name: DATABASE_NAME - value: litellm -{{- else if .Values.db.useExisting }} -- name: DATABASE_USERNAME - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.usernameKey }} -- name: DATABASE_PASSWORD - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.passwordKey }} -- name: DATABASE_HOST - {{- if .Values.db.secret.endpointKey }} - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.endpointKey }} - {{- else }} - value: {{ .Values.db.endpoint }} - {{- end }} -- name: DATABASE_NAME - value: {{ .Values.db.database }} -- name: DATABASE_URL - value: {{ .Values.db.url | quote }} -{{- end }} -{{- if and .Values.db.useExisting .Values.db.readReplicaUrl .Values.db.secret.readReplicaEndpointKey (not .Values.db.secret.readReplicaUrlKey) }} -- name: DATABASE_READER_HOST - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.readReplicaEndpointKey }} -{{- end }} -{{- if and .Values.db.useExisting .Values.db.secret.readReplicaUrlKey }} -- name: DATABASE_URL_READ_REPLICA - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.readReplicaUrlKey }} -{{- else if .Values.db.readReplicaUrl }} -- name: DATABASE_URL_READ_REPLICA - value: {{ .Values.db.readReplicaUrl | quote }} -{{- end }} -{{- if .Values.db.connectionPool.enabled }} -- name: LITELLM_PGBOUNCER_ENABLED - value: "true" -- name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: {{ .Values.db.connectionPool.maxDbConnections | quote }} -- name: LITELLM_PGBOUNCER_MAX_CLIENT_CONN - value: {{ .Values.db.connectionPool.maxClientConn | quote }} -{{- end }} -- name: PROXY_MASTER_KEY - valueFrom: - secretKeyRef: - name: {{ .Values.masterkeySecretName | default (printf "%s-masterkey" (include "litellm.fullname" .)) }} - key: {{ .Values.masterkeySecretKey | default "masterkey" }} -{{- if .Values.redis.enabled }} -- name: REDIS_HOST - value: {{ include "litellm.redis.serviceName" . }} -- name: REDIS_PORT - value: {{ include "litellm.redis.port" . | quote }} -- name: REDIS_PASSWORD - valueFrom: - secretKeyRef: - name: {{ include "redis.secretName" .Subcharts.redis }} - key: {{include "redis.secretPasswordKey" .Subcharts.redis }} -{{- end }} -{{- /* - Inject LITELLM_LOG only when envVars does not already define it. -*/}} -{{- if and .Values.logLevel (not (hasKey (default dict .Values.envVars) "LITELLM_LOG")) }} -- name: LITELLM_LOG - value: {{ .Values.logLevel | quote }} -{{- end }} -{{- if .Values.envVars }} -{{- range $key, $val := .Values.envVars }} -- name: {{ $key }} - value: {{ $val | quote }} -{{- end }} -{{- end }} -{{- with .Values.extraEnvVars }} -{{ toYaml . }} -{{- end }} -{{- if .Values.migrationJob.enabled }} -# Schema updates are owned by the dedicated migrations Job; skip -# the proxy's startup `prisma db push` so N replicas don't race -# one DB on every rollout. Placed last (after envVars and -# extraEnvVars) so this override can't be silently shadowed by a -# user-supplied DISABLE_SCHEMA_UPDATE under last-wins duplicate-env -# semantics — same pattern the migrations Job uses. -- name: DISABLE_SCHEMA_UPDATE - value: "true" -{{- end }} -{{- end -}} - -{{/* -Proxy-only metering and metrics env. The collector sidecar serves no HTTP -traffic, so it gets neither. -*/}} -{{- define "litellm.proxyMetricsEnv" -}} -{{- if .Values.billingMetrics.enabled }} -{{ include "litellm.billingMetricsEnv" . }} -{{- end }} -{{- if .Values.metricsServer.enabled }} -{{- if eq (int .Values.metricsServer.port) (int .Values.service.port) }} -{{- fail "metricsServer.port must differ from service.port" }} -{{- end }} -- name: PROMETHEUS_METRICS_PORT - value: {{ .Values.metricsServer.port | quote }} -{{- end }} -{{- end -}} - -{{/* -Directory of the collector's unix socket, shared between the two containers -through an emptyDir. Empty when the sidecar is off or uses 127.0.0.1 TCP. -*/}} -{{- define "litellm.collector.socketDir" -}} -{{- if and .Values.collector.enabled (hasPrefix "unix://" .Values.collector.address) -}} -{{- dir (trimPrefix "unix://" .Values.collector.address) -}} -{{- end -}} -{{- end -}} - -{{- define "litellm.collectorEnv" -}} -- name: LITELLM_COLLECTOR_ENABLED - value: "true" -- name: LITELLM_COLLECTOR_ADDRESS - value: {{ .Values.collector.address | quote }} -- name: LITELLM_COLLECTOR_BUFFER_SIZE - value: {{ .Values.collector.bufferSize | quote }} -- name: LITELLM_COLLECTOR_ON_UNAVAILABLE - value: {{ .Values.collector.onUnavailable | quote }} -- name: LITELLM_COLLECTOR_DRAIN_TIMEOUT_SECONDS - value: {{ .Values.collector.drainTimeoutSeconds | quote }} -{{- end -}} diff --git a/helm/litellm-helm/templates/configmap-litellm.yaml b/helm/litellm-helm/templates/configmap-litellm.yaml deleted file mode 100644 index 03e4f620206..00000000000 --- a/helm/litellm-helm/templates/configmap-litellm.yaml +++ /dev/null @@ -1,22 +0,0 @@ -{{- if .Values.proxyConfigMap.create }} -{{- $config := deepCopy .Values.proxy_config }} -{{- if and .Values.redis.enabled (dig "coordination" "enabled" true .Values.redis) }} -{{- $generalSettings := (get $config "general_settings") | default dict }} -{{- if not (hasKey $generalSettings "coordination_redis") }} -{{- $coordinationRedis := dict "host" "os.environ/REDIS_HOST" "port" "os.environ/REDIS_PORT" "password" "os.environ/REDIS_PASSWORD" }} -{{- if .Values.redis.sentinel.enabled }} -{{- $sentinelNode := list (include "litellm.redis.serviceName" .) (include "litellm.redis.port" . | int) }} -{{- $coordinationRedis = dict "sentinel_nodes" (list $sentinelNode) "service_name" (default "mymaster" .Values.redis.sentinel.masterSet) "password" "os.environ/REDIS_PASSWORD" }} -{{- end }} -{{- $_ := set $generalSettings "coordination_redis" $coordinationRedis }} -{{- $_ := set $config "general_settings" $generalSettings }} -{{- end }} -{{- end }} -apiVersion: v1 -kind: ConfigMap -metadata: - name: {{ include "litellm.fullname" . }}-config -data: - config.yaml: | -{{ $config | toYaml | indent 6 }} -{{- end }} diff --git a/helm/litellm-helm/templates/deployment.yaml b/helm/litellm-helm/templates/deployment.yaml deleted file mode 100644 index cf7b3f8a38d..00000000000 --- a/helm/litellm-helm/templates/deployment.yaml +++ /dev/null @@ -1,250 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: - {{- toYaml .Values.deploymentAnnotations | nindent 4 }} - name: {{ include "litellm.fullname" . }} - labels: - {{- include "litellm.labels" . | nindent 4 }} - {{- if .Values.deploymentLabels }} - {{- toYaml .Values.deploymentLabels | nindent 4 }} - {{- end }} -spec: - {{- if and (not .Values.keda.enabled) (not .Values.autoscaling.enabled) }} - replicas: {{ .Values.replicaCount }} - {{- end }} - {{- with .Values.strategy }} - strategy: - {{- toYaml . | nindent 4 }} - {{- end }} - selector: - matchLabels: - {{- include "litellm.selectorLabels" . | nindent 6 }} - {{- if .Values.deploymentMinReadySeconds }} - minReadySeconds: {{ .Values.deploymentMinReadySeconds }} - {{- end }} - template: - metadata: - annotations: - {{- if .Values.proxyConfigMap.create }} - checksum/config: {{ include (print $.Template.BasePath "/configmap-litellm.yaml") . | sha256sum }} - {{- end }} - {{- with .Values.podAnnotations }} - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - labels: - {{- include "litellm.labels" . | nindent 8 }} - {{- with .Values.podLabels }} - {{- toYaml . | nindent 8 }} - {{- end }} - spec: - {{- with .Values.imagePullSecrets }} - imagePullSecrets: - {{- toYaml . | nindent 8 }} - {{- end }} - serviceAccountName: {{ include "litellm.serviceAccountName" . }} - securityContext: - {{- toYaml .Values.podSecurityContext | nindent 8 }} - {{- with .Values.extraInitContainers }} - initContainers: - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - containers: - - name: {{ include "litellm.name" . }} - securityContext: - {{- toYaml .Values.securityContext | nindent 12 }} - image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.image.pullPolicy }} - env: - {{- include "litellm.proxyEnv" . | nindent 12 }} - {{- include "litellm.proxyMetricsEnv" . | nindent 12 }} - {{- if .Values.collector.enabled }} - {{- include "litellm.collectorEnv" . | nindent 12 }} - {{- end }} - envFrom: - {{- range .Values.environmentSecrets }} - - secretRef: - name: {{ . }} - {{- end }} - {{- range .Values.environmentConfigMaps }} - - configMapRef: - name: {{ . }} - {{- end }} - {{- if .Values.command }} - command: {{ toYaml .Values.command | nindent 12 }} - {{- end }} - {{- if .Values.args }} - args: {{ toYaml .Values.args | nindent 12 }} - {{- else }} - args: - - --config - - /etc/litellm/config.yaml - {{ if .Values.numWorkers }} - - --num_workers - - {{ .Values.numWorkers | quote }} - {{- end }} - {{- end }} - ports: - - name: http - containerPort: {{ .Values.service.port }} - protocol: TCP - {{- if .Values.metricsServer.enabled }} - - name: metrics - containerPort: {{ .Values.metricsServer.port }} - protocol: TCP - {{- end }} - livenessProbe: - httpGet: - path: {{ .Values.livenessProbe.path | quote }} - port: "http" - initialDelaySeconds: {{ .Values.livenessProbe.initialDelaySeconds }} - periodSeconds: {{ .Values.livenessProbe.periodSeconds }} - timeoutSeconds: {{ .Values.livenessProbe.timeoutSeconds }} - successThreshold: {{ .Values.livenessProbe.successThreshold }} - failureThreshold: {{ .Values.livenessProbe.failureThreshold }} - readinessProbe: - httpGet: - path: {{ .Values.readinessProbe.path | quote }} - port: "http" - initialDelaySeconds: {{ .Values.readinessProbe.initialDelaySeconds }} - periodSeconds: {{ .Values.readinessProbe.periodSeconds }} - timeoutSeconds: {{ .Values.readinessProbe.timeoutSeconds }} - successThreshold: {{ .Values.readinessProbe.successThreshold }} - failureThreshold: {{ .Values.readinessProbe.failureThreshold }} - startupProbe: - httpGet: - path: {{ .Values.startupProbe.path | quote }} - port: "http" - initialDelaySeconds: {{ .Values.startupProbe.initialDelaySeconds }} - periodSeconds: {{ .Values.startupProbe.periodSeconds }} - timeoutSeconds: {{ .Values.startupProbe.timeoutSeconds }} - successThreshold: {{ .Values.startupProbe.successThreshold }} - failureThreshold: {{ .Values.startupProbe.failureThreshold }} - resources: - {{- toYaml .Values.resources | nindent 12 }} - volumeMounts: - - name: litellm-config - mountPath: /etc/litellm/config.yaml - subPath: config.yaml - {{ if .Values.securityContext.readOnlyRootFilesystem }} - - name: tmp - mountPath: /tmp - - name: cache - mountPath: /.cache - - name: npm - mountPath: /.npm - {{- end }} - {{- if .Values.billingMetrics.enabled }} - {{- include "litellm.billingMetricsVolumeMounts" . | nindent 12 }} - {{- end }} - {{- if include "litellm.collector.socketDir" . }} - - name: collector-socket - mountPath: {{ include "litellm.collector.socketDir" . }} - {{- end }} - {{- with .Values.volumeMounts }} - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.lifecycle }} - lifecycle: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- if .Values.collector.enabled }} - - name: {{ include "litellm.name" . }}-collector - securityContext: - {{- toYaml .Values.securityContext | nindent 12 }} - image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.image.pullPolicy }} - command: {{ toYaml .Values.collector.command | nindent 12 }} - env: - {{- include "litellm.proxyEnv" . | nindent 12 }} - {{- include "litellm.collectorEnv" . | nindent 12 }} - - name: LITELLM_JOB_ROLE - value: collector - {{- if not (hasKey (default dict .Values.envVars) "CONFIG_FILE_PATH") }} - - name: CONFIG_FILE_PATH - value: /etc/litellm/config.yaml - {{- end }} - envFrom: - {{- range .Values.environmentSecrets }} - - secretRef: - name: {{ . }} - {{- end }} - {{- range .Values.environmentConfigMaps }} - - configMapRef: - name: {{ . }} - {{- end }} - resources: - {{- toYaml .Values.collector.resources | nindent 12 }} - volumeMounts: - - name: litellm-config - mountPath: /etc/litellm/config.yaml - subPath: config.yaml - {{- if include "litellm.collector.socketDir" . }} - - name: collector-socket - mountPath: {{ include "litellm.collector.socketDir" . }} - {{- end }} - {{ if .Values.securityContext.readOnlyRootFilesystem }} - - name: tmp - mountPath: /tmp - - name: cache - mountPath: /.cache - - name: npm - mountPath: /.npm - {{- end }} - {{- with .Values.volumeMounts }} - {{- toYaml . | nindent 12 }} - {{- end }} - {{- end }} - {{- with .Values.extraContainers }} - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - volumes: - {{ if .Values.securityContext.readOnlyRootFilesystem }} - - name: tmp - emptyDir: - sizeLimit: 500Mi - - name: cache - emptyDir: - sizeLimit: 500Mi - - name: npm - emptyDir: - sizeLimit: 500Mi - {{- end }} - - name: litellm-config - configMap: - {{- if .Values.proxyConfigMap.create }} - name: {{ include "litellm.fullname" . }}-config - {{- else }} - name: {{ .Values.proxyConfigMap.name }} - {{- end }} - items: - - key: {{ .Values.proxyConfigMap.key | default "config.yaml" }} - path: "config.yaml" - {{- if .Values.billingMetrics.enabled }} - {{- include "litellm.billingMetricsVolumes" . | nindent 8 }} - {{- end }} - {{- if include "litellm.collector.socketDir" . }} - - name: collector-socket - emptyDir: - sizeLimit: 1Mi - {{- end }} - {{- with .Values.volumes }} - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.nodeSelector }} - nodeSelector: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.affinity }} - affinity: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.tolerations }} - tolerations: - {{- toYaml . | nindent 8 }} - {{- end }} - terminationGracePeriodSeconds: {{ .Values.terminationGracePeriodSeconds | default 90 }} - {{- if .Values.topologySpreadConstraints }} - topologySpreadConstraints: - {{- toYaml .Values.topologySpreadConstraints | nindent 8 }} - {{- end }} diff --git a/helm/litellm-helm/templates/extra-resources.yaml b/helm/litellm-helm/templates/extra-resources.yaml deleted file mode 100644 index 33190d96fc0..00000000000 --- a/helm/litellm-helm/templates/extra-resources.yaml +++ /dev/null @@ -1,6 +0,0 @@ -{{- if .Values.extraResources }} -{{- range .Values.extraResources }} ---- -{{ toYaml . | nindent 0 }} -{{- end }} -{{- end }} \ No newline at end of file diff --git a/helm/litellm-helm/templates/hpa.yaml b/helm/litellm-helm/templates/hpa.yaml deleted file mode 100644 index a651f916d21..00000000000 --- a/helm/litellm-helm/templates/hpa.yaml +++ /dev/null @@ -1,64 +0,0 @@ -{{- if .Values.autoscaling.enabled }} -apiVersion: autoscaling/v2 -kind: HorizontalPodAutoscaler -metadata: - name: {{ include "litellm.fullname" . }} - labels: - {{- include "litellm.labels" . | nindent 4 }} -spec: - scaleTargetRef: - apiVersion: apps/v1 - kind: Deployment - name: {{ include "litellm.fullname" . }} - minReplicas: {{ .Values.autoscaling.minReplicas }} - maxReplicas: {{ .Values.autoscaling.maxReplicas }} - {{- if .Values.autoscaling.behavior }} - behavior: - {{- toYaml .Values.autoscaling.behavior | nindent 4 }} - {{- end }} - metrics: - {{- if .Values.autoscaling.targetCPUUtilizationPercentage }} - {{- if and .Values.collector.enabled .Values.collector.scaleOnProxyContainerCpu }} - - type: ContainerResource - containerResource: - name: cpu - container: {{ include "litellm.name" . }} - target: - type: Utilization - averageUtilization: {{ .Values.autoscaling.targetCPUUtilizationPercentage }} - {{- else }} - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: {{ .Values.autoscaling.targetCPUUtilizationPercentage }} - {{- end }} - {{- end }} - {{- if .Values.autoscaling.targetMemoryUtilizationPercentage }} - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: {{ .Values.autoscaling.targetMemoryUtilizationPercentage }} - {{- end }} - {{- with .Values.autoscaling.targetRequestsPerSecond }} - - type: Pods - pods: - metric: - name: litellm_requests_per_second - target: - type: AverageValue - averageValue: {{ toJson . | trimAll "\"" | quote }} - {{- end }} - {{- with .Values.autoscaling.targetTokensPerSecond }} - - type: Pods - pods: - metric: - name: litellm_tokens_per_second - target: - type: AverageValue - averageValue: {{ toJson . | trimAll "\"" | quote }} - {{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/ingress.yaml b/helm/litellm-helm/templates/ingress.yaml deleted file mode 100644 index ea9ffcbb54c..00000000000 --- a/helm/litellm-helm/templates/ingress.yaml +++ /dev/null @@ -1,64 +0,0 @@ -{{- if .Values.ingress.enabled -}} -{{- $fullName := include "litellm.fullname" . -}} -{{- $svcPort := .Values.service.port -}} -{{- if and .Values.ingress.className (not (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion)) }} - {{- if not (hasKey .Values.ingress.annotations "kubernetes.io/ingress.class") }} - {{- $_ := set .Values.ingress.annotations "kubernetes.io/ingress.class" .Values.ingress.className}} - {{- end }} -{{- end }} -{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion -}} -apiVersion: networking.k8s.io/v1 -{{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion -}} -apiVersion: networking.k8s.io/v1beta1 -{{- else -}} -apiVersion: extensions/v1beta1 -{{- end }} -kind: Ingress -metadata: - name: {{ $fullName }} - labels: - {{- include "litellm.labels" . | nindent 4 }} - {{- with .Values.ingress.labels }} - {{- toYaml . | nindent 4 }} - {{- end }} - {{- with .Values.ingress.annotations }} - annotations: - {{- toYaml . | nindent 4 }} - {{- end }} -spec: - {{- if and .Values.ingress.className (semverCompare ">=1.18-0" .Capabilities.KubeVersion.GitVersion) }} - ingressClassName: {{ .Values.ingress.className }} - {{- end }} - {{- if .Values.ingress.tls }} - tls: - {{- range .Values.ingress.tls }} - - hosts: - {{- range .hosts }} - - {{ . | quote }} - {{- end }} - secretName: {{ .secretName }} - {{- end }} - {{- end }} - rules: - {{- range .Values.ingress.hosts }} - - host: {{ .host | quote }} - http: - paths: - {{- range .paths }} - - path: {{ .path }} - {{- if and .pathType (semverCompare ">=1.18-0" $.Capabilities.KubeVersion.GitVersion) }} - pathType: {{ .pathType }} - {{- end }} - backend: - {{- if semverCompare ">=1.19-0" $.Capabilities.KubeVersion.GitVersion }} - service: - name: {{ $fullName }} - port: - number: {{ $svcPort }} - {{- else }} - serviceName: {{ $fullName }} - servicePort: {{ $svcPort }} - {{- end }} - {{- end }} - {{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/keda.yaml b/helm/litellm-helm/templates/keda.yaml deleted file mode 100644 index bf585d0d4be..00000000000 --- a/helm/litellm-helm/templates/keda.yaml +++ /dev/null @@ -1,58 +0,0 @@ -{{- if and .Values.keda.enabled (not .Values.autoscaling.enabled) }} -apiVersion: keda.sh/v1alpha1 -kind: ScaledObject -metadata: - name: {{ include "litellm.fullname" . }} - labels: - {{- include "litellm.labels" . | nindent 4 }} - {{- if .Values.keda.scaledObject.annotations }} - annotations: {{ toYaml .Values.keda.scaledObject.annotations | nindent 4 }} - {{- end }} -spec: - scaleTargetRef: - name: {{ include "litellm.fullname" . }} - pollingInterval: {{ .Values.keda.pollingInterval }} - cooldownPeriod: {{ .Values.keda.cooldownPeriod }} - minReplicaCount: {{ .Values.keda.minReplicas }} - maxReplicaCount: {{ .Values.keda.maxReplicas }} -{{- with .Values.keda.fallback }} - fallback: - failureThreshold: {{ .failureThreshold | default 3 }} - replicas: {{ .replicas | default $.Values.keda.maxReplicas }} -{{- end }} - triggers: -{{- with .Values.keda.triggers }} - {{- toYaml . | nindent 2 }} -{{- end }} -{{- $prom := .Values.keda.prometheus }} -{{- if or $prom.requestsPerSecond $prom.tokensPerSecond }} -{{- if not $prom.serverAddress }} -{{- fail "keda.prometheus.serverAddress is required when keda.prometheus.requestsPerSecond or tokensPerSecond is set" }} -{{- end }} -{{- $selector := printf "namespace=%q,job=%q" .Release.Namespace (printf "%s%s" (include "litellm.fullname" .) (ternary "-metrics" "" .Values.metricsServer.enabled)) }} -{{- with $prom.requestsPerSecond }} - - type: prometheus - metadata: - serverAddress: {{ $prom.serverAddress | quote }} - threshold: {{ toJson . | trimAll "\"" | quote }} - query: {{ printf "sum(rate(litellm_proxy_total_requests_metric_total{%s}[1m]))" $selector | quote }} -{{- end }} -{{- with $prom.tokensPerSecond }} - - type: prometheus - metadata: - serverAddress: {{ $prom.serverAddress | quote }} - threshold: {{ toJson . | trimAll "\"" | quote }} - query: {{ printf "sum(rate(litellm_total_tokens_metric_total{%s}[1m]))" $selector | quote }} -{{- end }} -{{- end }} - advanced: - restoreToOriginalReplicaCount: {{ .Values.keda.restoreToOriginalReplicaCount }} -{{- if .Values.keda.behavior }} - horizontalPodAutoscalerConfig: - behavior: -{{- with .Values.keda.behavior }} -{{- toYaml . | nindent 8 }} -{{- end }} - -{{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/migrations-job.yaml b/helm/litellm-helm/templates/migrations-job.yaml deleted file mode 100644 index 5a873cbb965..00000000000 --- a/helm/litellm-helm/templates/migrations-job.yaml +++ /dev/null @@ -1,125 +0,0 @@ -{{- if .Values.migrationJob.enabled }} -# This job runs the Prisma migrations for the LiteLLM DB. -apiVersion: batch/v1 -kind: Job -metadata: - name: {{ include "litellm.fullname" . }}-migrations - labels: - {{- include "litellm.labels" . | nindent 4 }} - annotations: - {{- if .Values.migrationJob.hooks.argocd.enabled }} - argocd.argoproj.io/hook: PreSync - argocd.argoproj.io/hook-delete-policy: BeforeHookCreation - {{- end }} - {{- if .Values.migrationJob.hooks.helm.enabled }} - helm.sh/hook: "pre-install,pre-upgrade" - helm.sh/hook-delete-policy: "before-hook-creation" - helm.sh/hook-weight: {{ .Values.migrationJob.hooks.helm.weight | default "1" | quote }} - {{- end }} - checksum/config: {{ toYaml .Values | sha256sum }} -spec: - template: - metadata: - labels: - {{- include "litellm.labels" . | nindent 8 }} - {{- with .Values.podLabels }} - {{- toYaml . | nindent 8 }} - {{- end }} - annotations: - {{- with .Values.migrationJob.annotations }} - {{- toYaml . | nindent 8 }} - {{- end }} - spec: - {{- with .Values.imagePullSecrets }} - imagePullSecrets: - {{- toYaml . | nindent 8 }} - {{- end }} - serviceAccountName: {{ include "litellm.migrationServiceAccountName" . }} - securityContext: - {{- toYaml .Values.podSecurityContext | nindent 8 }} - {{- with .Values.migrationJob.extraInitContainers }} - initContainers: - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - containers: - - name: prisma-migrations - image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.image.pullPolicy }} - securityContext: - {{- toYaml .Values.securityContext | nindent 12 }} - command: ["python", "litellm/proxy/prisma_migration.py"] - workingDir: "/app" - env: - {{- if .Values.db.useExisting }} - - name: DATABASE_USERNAME - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.usernameKey }} - - name: DATABASE_PASSWORD - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.passwordKey }} - - name: DATABASE_HOST - {{- if .Values.db.secret.endpointKey }} - valueFrom: - secretKeyRef: - name: {{ .Values.db.secret.name }} - key: {{ .Values.db.secret.endpointKey }} - {{- else }} - value: {{ .Values.db.endpoint }} - {{- end }} - - name: DATABASE_NAME - value: {{ .Values.db.database }} - - name: DATABASE_URL - value: {{ .Values.db.url | quote }} - {{- else if .Values.db.deployStandalone }} - - name: DATABASE_URL - value: postgresql://{{ .Values.postgresql.auth.username }}:{{ .Values.postgresql.auth.password }}@{{ .Release.Name }}-postgresql/{{ .Values.postgresql.auth.database }} - {{- end }} - {{- if .Values.envVars }} - {{- range $key, $val := .Values.envVars }} - - name: {{ $key }} - value: {{ $val | quote }} - {{- end }} - {{- end }} - {{- with .Values.extraEnvVars }} - {{- toYaml . | nindent 12 }} - {{- end }} - - name: DISABLE_SCHEMA_UPDATE - value: "false" # always run the migration from the Helm PreSync hook, override the value set - {{- with .Values.volumeMounts }} - volumeMounts: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.migrationJob.resources }} - resources: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.migrationJob.extraContainers }} - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - {{- with .Values.volumes }} - volumes: - {{- toYaml . | nindent 8 }} - {{- end }} - restartPolicy: OnFailure - {{- with .Values.nodeSelector }} - nodeSelector: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.affinity }} - affinity: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.tolerations }} - tolerations: - {{- toYaml . | nindent 8 }} - {{- end }} - ttlSecondsAfterFinished: {{ .Values.migrationJob.ttlSecondsAfterFinished }} - backoffLimit: {{ .Values.migrationJob.backoffLimit }} - {{- with .Values.migrationJob.activeDeadlineSeconds }} - activeDeadlineSeconds: {{ . }} - {{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/poddisruptionbudget.yaml b/helm/litellm-helm/templates/poddisruptionbudget.yaml deleted file mode 100644 index 1715b94c1f6..00000000000 --- a/helm/litellm-helm/templates/poddisruptionbudget.yaml +++ /dev/null @@ -1,33 +0,0 @@ -{{- /* -PodDisruptionBudget for LiteLLM proxy -Controlled via .Values.pdb.enabled and .Values.pdb.{minAvailable|maxUnavailable} -Only one of minAvailable / maxUnavailable should be set. If both are set, minAvailable wins. -*/ -}} -{{- if .Values.pdb.enabled }} -apiVersion: policy/v1 -kind: PodDisruptionBudget -metadata: - name: {{ include "litellm.fullname" . }} - labels: - {{- include "litellm.labels" . | nindent 4 }} - {{- with .Values.pdb.labels }} - {{- toYaml . | nindent 4 }} - {{- end }} - {{- with .Values.pdb.annotations }} - annotations: - {{- toYaml . | nindent 4 }} - {{- end }} -spec: - selector: - matchLabels: - {{- /* Match the Deployment selector to target the same pod set */ -}} - {{- include "litellm.selectorLabels" . | nindent 6 }} - {{- if .Values.pdb.minAvailable }} - minAvailable: {{ .Values.pdb.minAvailable }} - {{- else if .Values.pdb.maxUnavailable }} - maxUnavailable: {{ .Values.pdb.maxUnavailable }} - {{- else }} - # Safe default if enabled but not configured - maxUnavailable: 1 - {{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/secret-dbcredentials.yaml b/helm/litellm-helm/templates/secret-dbcredentials.yaml deleted file mode 100644 index 8ab89a4579e..00000000000 --- a/helm/litellm-helm/templates/secret-dbcredentials.yaml +++ /dev/null @@ -1,13 +0,0 @@ -{{- if .Values.db.deployStandalone -}} -{{- include "litellm.validateBundledPostgresImageTag" . -}} -apiVersion: v1 -kind: Secret -metadata: - name: {{ include "litellm.fullname" . }}-dbcredentials -data: - # Password for the "postgres" user - postgres-password: {{ ( index .Values.postgresql.auth "postgres-password") | default "litellm" | b64enc }} - username: {{ .Values.postgresql.auth.username | default "litellm" | b64enc }} - password: {{ .Values.postgresql.auth.password | default "litellm" | b64enc }} -type: Opaque -{{- end -}} \ No newline at end of file diff --git a/helm/litellm-helm/templates/secret-masterkey.yaml b/helm/litellm-helm/templates/secret-masterkey.yaml deleted file mode 100644 index 60ab4e74c6b..00000000000 --- a/helm/litellm-helm/templates/secret-masterkey.yaml +++ /dev/null @@ -1,12 +0,0 @@ -{{- if not .Values.masterkeySecretName }} -{{- $secretName := printf "%s-masterkey" (include "litellm.fullname" .) }} -{{- $existing := lookup "v1" "Secret" .Release.Namespace $secretName }} -{{- $masterkey := .Values.masterkey | default (dig "data" "masterkey" "" $existing | b64dec) | default (printf "sk-%s" (randAlphaNum 18)) }} -apiVersion: v1 -kind: Secret -metadata: - name: {{ $secretName }} -data: - masterkey: {{ $masterkey | b64enc }} -type: Opaque -{{- end }} diff --git a/helm/litellm-helm/templates/service-metrics.yaml b/helm/litellm-helm/templates/service-metrics.yaml deleted file mode 100644 index 1d23fe39606..00000000000 --- a/helm/litellm-helm/templates/service-metrics.yaml +++ /dev/null @@ -1,17 +0,0 @@ -{{- if .Values.metricsServer.enabled }} -apiVersion: v1 -kind: Service -metadata: - name: {{ include "litellm.fullname" . }}-metrics - labels: - {{- include "litellm.labels" . | nindent 4 }} -spec: - type: ClusterIP - ports: - - port: {{ .Values.metricsServer.port }} - targetPort: metrics - protocol: TCP - name: metrics - selector: - {{- include "litellm.selectorLabels" . | nindent 4 }} -{{- end }} diff --git a/helm/litellm-helm/templates/service.yaml b/helm/litellm-helm/templates/service.yaml deleted file mode 100644 index 11812208929..00000000000 --- a/helm/litellm-helm/templates/service.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: v1 -kind: Service -metadata: - name: {{ include "litellm.fullname" . }} - {{- with .Values.service.annotations }} - annotations: - {{- toYaml . | nindent 4 }} - {{- end }} - labels: - {{- include "litellm.labels" . | nindent 4 }} -spec: - type: {{ .Values.service.type }} - {{- if and (eq .Values.service.type "LoadBalancer") .Values.service.loadBalancerClass }} - loadBalancerClass: {{ .Values.service.loadBalancerClass }} - {{- end }} - ports: - - port: {{ .Values.service.port }} - targetPort: http - protocol: TCP - name: http - selector: - {{- include "litellm.selectorLabels" . | nindent 4 }} diff --git a/helm/litellm-helm/templates/serviceaccount.yaml b/helm/litellm-helm/templates/serviceaccount.yaml deleted file mode 100644 index 7655470fa42..00000000000 --- a/helm/litellm-helm/templates/serviceaccount.yaml +++ /dev/null @@ -1,13 +0,0 @@ -{{- if .Values.serviceAccount.create -}} -apiVersion: v1 -kind: ServiceAccount -metadata: - name: {{ include "litellm.serviceAccountName" . }} - labels: - {{- include "litellm.labels" . | nindent 4 }} - {{- with .Values.serviceAccount.annotations }} - annotations: - {{- toYaml . | nindent 4 }} - {{- end }} -automountServiceAccountToken: {{ .Values.serviceAccount.automount }} -{{- end }} diff --git a/helm/litellm-helm/templates/servicemonitor.yaml b/helm/litellm-helm/templates/servicemonitor.yaml deleted file mode 100644 index 68083d0da61..00000000000 --- a/helm/litellm-helm/templates/servicemonitor.yaml +++ /dev/null @@ -1,39 +0,0 @@ -{{- with .Values.serviceMonitor }} -{{- if and (eq .enabled true) }} -apiVersion: monitoring.coreos.com/v1 -kind: ServiceMonitor -metadata: - name: {{ include "litellm.fullname" $ }} - labels: - {{- include "litellm.labels" $ | nindent 4 }} - {{- if .labels }} - {{- toYaml .labels | nindent 4 }} - {{- end }} - {{- if .annotations }} - annotations: - {{- toYaml .annotations | nindent 4 }} - {{- end }} -spec: - selector: - matchLabels: - {{- include "litellm.selectorLabels" $ | nindent 6 }} - namespaceSelector: - matchNames: - # if not set, use the release namespace - {{- if not .namespaceSelector.matchNames }} - - {{ $.Release.Namespace | quote }} - {{- else }} - {{- toYaml .namespaceSelector.matchNames | nindent 4 }} - {{- end }} - endpoints: - - port: {{ ternary "metrics" "http" $.Values.metricsServer.enabled }} - path: /metrics/ - interval: {{ .interval }} - scrapeTimeout: {{ .scrapeTimeout }} - scheme: http - {{- if .relabelings }} - relabelings: -{{- toYaml .relabelings | nindent 4 }} - {{- end }} -{{- end }} -{{- end }} diff --git a/helm/litellm-helm/templates/tests/test-connection.yaml b/helm/litellm-helm/templates/tests/test-connection.yaml deleted file mode 100644 index 86a8f66b10b..00000000000 --- a/helm/litellm-helm/templates/tests/test-connection.yaml +++ /dev/null @@ -1,25 +0,0 @@ -apiVersion: v1 -kind: Pod -metadata: - name: "{{ include "litellm.fullname" . }}-test-connection" - labels: - {{- include "litellm.labels" . | nindent 4 }} - annotations: - "helm.sh/hook": test -spec: - containers: - - name: wget - image: busybox - command: ['sh', '-c'] - args: - - | - # Wait for a bit to allow the service to be ready - sleep 10 - # Try multiple times with a delay between attempts - for i in $(seq 1 30); do - wget -T 5 "{{ include "litellm.fullname" . }}:{{ .Values.service.port }}/health/readiness" && exit 0 - echo "Attempt $i failed, waiting..." - sleep 2 - done - exit 1 - restartPolicy: Never \ No newline at end of file diff --git a/helm/litellm-helm/templates/tests/test-env-vars.yaml b/helm/litellm-helm/templates/tests/test-env-vars.yaml deleted file mode 100644 index 9f0277557a4..00000000000 --- a/helm/litellm-helm/templates/tests/test-env-vars.yaml +++ /dev/null @@ -1,43 +0,0 @@ -apiVersion: v1 -kind: Pod -metadata: - name: "{{ include "litellm.fullname" . }}-env-test" - labels: - {{- include "litellm.labels" . | nindent 4 }} - annotations: - "helm.sh/hook": test -spec: - containers: - - name: test - image: busybox - command: ['sh', '-c'] - args: - - | - # Test DD_ENV - if [ "$DD_ENV" != "dev_helm" ]; then - echo "❌ Environment variable DD_ENV mismatch. Expected: dev_helm, Got: $DD_ENV" - exit 1 - fi - echo "✅ Environment variable DD_ENV matches expected value: $DD_ENV" - - # Test DD_SERVICE - if [ "$DD_SERVICE" != "litellm" ]; then - echo "❌ Environment variable DD_SERVICE mismatch. Expected: litellm, Got: $DD_SERVICE" - exit 1 - fi - echo "✅ Environment variable DD_SERVICE matches expected value: $DD_SERVICE" - - # Test USE_DDTRACE - if [ "$USE_DDTRACE" != "true" ]; then - echo "❌ Environment variable USE_DDTRACE mismatch. Expected: true, Got: $USE_DDTRACE" - exit 1 - fi - echo "✅ Environment variable USE_DDTRACE matches expected value: $USE_DDTRACE" - env: - - name: DD_ENV - value: {{ .Values.envVars.DD_ENV | quote }} - - name: DD_SERVICE - value: {{ .Values.envVars.DD_SERVICE | quote }} - - name: USE_DDTRACE - value: {{ .Values.envVars.USE_DDTRACE | quote }} - restartPolicy: Never \ No newline at end of file diff --git a/helm/litellm-helm/templates/tests/test-servicemonitor.yaml b/helm/litellm-helm/templates/tests/test-servicemonitor.yaml deleted file mode 100644 index ef8475339c3..00000000000 --- a/helm/litellm-helm/templates/tests/test-servicemonitor.yaml +++ /dev/null @@ -1,152 +0,0 @@ -{{- if .Values.serviceMonitor.enabled }} -apiVersion: v1 -kind: Pod -metadata: - name: "{{ include "litellm.fullname" . }}-test-servicemonitor" - labels: - {{- include "litellm.labels" . | nindent 4 }} - annotations: - "helm.sh/hook": test -spec: - containers: - - name: test - image: docker.io/bitnamilegacy/kubectl:1.29.2-debian-12-r3 - command: ['sh', '-c'] - args: - - | - set -e - echo "🔍 Testing ServiceMonitor configuration..." - - # Check if ServiceMonitor exists - if ! kubectl get servicemonitor {{ include "litellm.fullname" . }} -n {{ .Release.Namespace }} &>/dev/null; then - echo "❌ ServiceMonitor not found" - exit 1 - fi - echo "✅ ServiceMonitor exists" - - # Get ServiceMonitor YAML - SM=$(kubectl get servicemonitor {{ include "litellm.fullname" . }} -n {{ .Release.Namespace }} -o yaml) - - # Test endpoint configuration - ENDPOINT_PORT=$(echo "$SM" | grep -A 5 "endpoints:" | grep "port:" | awk '{print $2}') - if [ "$ENDPOINT_PORT" != "http" ]; then - echo "❌ Endpoint port mismatch. Expected: http, Got: $ENDPOINT_PORT" - exit 1 - fi - echo "✅ Endpoint port is correctly set to: $ENDPOINT_PORT" - - # Test endpoint path - ENDPOINT_PATH=$(echo "$SM" | grep -A 5 "endpoints:" | grep "path:" | awk '{print $2}') - if [ "$ENDPOINT_PATH" != "/metrics/" ]; then - echo "❌ Endpoint path mismatch. Expected: /metrics/, Got: $ENDPOINT_PATH" - exit 1 - fi - echo "✅ Endpoint path is correctly set to: $ENDPOINT_PATH" - - # Test interval - INTERVAL=$(echo "$SM" | grep "interval:" | awk '{print $2}') - if [ "$INTERVAL" != "{{ .Values.serviceMonitor.interval }}" ]; then - echo "❌ Interval mismatch. Expected: {{ .Values.serviceMonitor.interval }}, Got: $INTERVAL" - exit 1 - fi - echo "✅ Interval is correctly set to: $INTERVAL" - - # Test scrapeTimeout - TIMEOUT=$(echo "$SM" | grep "scrapeTimeout:" | awk '{print $2}') - if [ "$TIMEOUT" != "{{ .Values.serviceMonitor.scrapeTimeout }}" ]; then - echo "❌ ScrapeTimeout mismatch. Expected: {{ .Values.serviceMonitor.scrapeTimeout }}, Got: $TIMEOUT" - exit 1 - fi - echo "✅ ScrapeTimeout is correctly set to: $TIMEOUT" - - # Test scheme - SCHEME=$(echo "$SM" | grep "scheme:" | awk '{print $2}') - if [ "$SCHEME" != "http" ]; then - echo "❌ Scheme mismatch. Expected: http, Got: $SCHEME" - exit 1 - fi - echo "✅ Scheme is correctly set to: $SCHEME" - - {{- if .Values.serviceMonitor.labels }} - # Test custom labels - echo "🔍 Checking custom labels..." - {{- range $key, $value := .Values.serviceMonitor.labels }} - LABEL_VALUE=$(echo "$SM" | grep -A 20 "metadata:" | grep "{{ $key }}:" | awk '{print $2}') - if [ "$LABEL_VALUE" != "{{ $value }}" ]; then - echo "❌ Label {{ $key }} mismatch. Expected: {{ $value }}, Got: $LABEL_VALUE" - exit 1 - fi - echo "✅ Label {{ $key }} is correctly set to: {{ $value }}" - {{- end }} - {{- end }} - - {{- if .Values.serviceMonitor.annotations }} - # Test annotations - echo "🔍 Checking annotations..." - {{- range $key, $value := .Values.serviceMonitor.annotations }} - ANNOTATION_VALUE=$(echo "$SM" | grep -A 10 "annotations:" | grep "{{ $key }}:" | awk '{print $2}') - if [ "$ANNOTATION_VALUE" != "{{ $value }}" ]; then - echo "❌ Annotation {{ $key }} mismatch. Expected: {{ $value }}, Got: $ANNOTATION_VALUE" - exit 1 - fi - echo "✅ Annotation {{ $key }} is correctly set to: {{ $value }}" - {{- end }} - {{- end }} - - {{- if .Values.serviceMonitor.namespaceSelector.matchNames }} - # Test namespace selector - echo "🔍 Checking namespace selector..." - {{- range .Values.serviceMonitor.namespaceSelector.matchNames }} - if ! echo "$SM" | grep -A 5 "namespaceSelector:" | grep -q "{{ . }}"; then - echo "❌ Namespace {{ . }} not found in namespaceSelector" - exit 1 - fi - echo "✅ Namespace {{ . }} found in namespaceSelector" - {{- end }} - {{- else }} - # Test default namespace selector (should be release namespace) - if ! echo "$SM" | grep -A 5 "namespaceSelector:" | grep -q "{{ .Release.Namespace }}"; then - echo "❌ Release namespace {{ .Release.Namespace }} not found in namespaceSelector" - exit 1 - fi - echo "✅ Default namespace selector set to release namespace: {{ .Release.Namespace }}" - {{- end }} - - {{- if .Values.serviceMonitor.relabelings }} - # Test relabelings - echo "🔍 Checking relabelings configuration..." - if ! echo "$SM" | grep -q "relabelings:"; then - echo "❌ Relabelings section not found" - exit 1 - fi - echo "✅ Relabelings section exists" - {{- range .Values.serviceMonitor.relabelings }} - {{- if .targetLabel }} - if ! echo "$SM" | grep -A 50 "relabelings:" | grep -q "targetLabel: {{ .targetLabel }}"; then - echo "❌ Relabeling targetLabel {{ .targetLabel }} not found" - exit 1 - fi - echo "✅ Relabeling targetLabel {{ .targetLabel }} found" - {{- end }} - {{- if .action }} - if ! echo "$SM" | grep -A 50 "relabelings:" | grep -q "action: {{ .action }}"; then - echo "❌ Relabeling action {{ .action }} not found" - exit 1 - fi - echo "✅ Relabeling action {{ .action }} found" - {{- end }} - {{- end }} - {{- end }} - - # Test selector labels match the service - echo "🔍 Checking selector labels match service..." - SVC_LABELS=$(kubectl get svc {{ include "litellm.fullname" . }} -n {{ .Release.Namespace }} -o jsonpath='{.metadata.labels}') - echo "Service labels: $SVC_LABELS" - echo "✅ Selector labels validation passed" - - echo "" - echo "🎉 All ServiceMonitor tests passed successfully!" - serviceAccountName: {{ include "litellm.serviceAccountName" . }} - restartPolicy: Never -{{- end }} - diff --git a/helm/litellm-helm/tests/billing_metrics_tests.yaml b/helm/litellm-helm/tests/billing_metrics_tests.yaml deleted file mode 100644 index 71803df378c..00000000000 --- a/helm/litellm-helm/tests/billing_metrics_tests.yaml +++ /dev/null @@ -1,297 +0,0 @@ -suite: test billingMetrics wiring on the proxy deployment -templates: - - deployment.yaml - - configmap-litellm.yaml - - migrations-job.yaml -tests: - - it: is off by default, adding no env, volume, or mount - template: deployment.yaml - asserts: - - notContains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls - secret: - secretName: litellm-billing-metrics-mtls - - notContains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: billing-metrics-mtls - mountPath: /etc/litellm/billing-mtls - readOnly: true - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://telemetry.litellm.ai - - - it: renders the endpoint and the mounted cert paths when enabled - template: deployment.yaml - set: - billingMetrics: - enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://telemetry.litellm.ai - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_CLIENT_CERT - value: /etc/litellm/billing-mtls/tls.crt - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_CLIENT_KEY - value: /etc/litellm/billing-mtls/tls.key - - # The conventional Secret name is the default, so enabling the block is enough. - - it: mounts the default cert secret read-only alongside the config volume - template: deployment.yaml - set: - billingMetrics: - enabled: true - asserts: - - contains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls - secret: - secretName: litellm-billing-metrics-mtls - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: billing-metrics-mtls - mountPath: /etc/litellm/billing-mtls - readOnly: true - - - it: honours a secretName override - template: deployment.yaml - set: - billingMetrics: - enabled: true - secretName: my-billing-mtls - asserts: - - contains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls - secret: - secretName: my-billing-mtls - - notContains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls - secret: - secretName: litellm-billing-metrics-mtls - - - it: honours an endpoint override - template: deployment.yaml - set: - billingMetrics: - enabled: true - endpoint: https://collector.internal:4318 - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://collector.internal:4318 - - # The production collector presents a public web-PKI certificate, so the CA - # override must stay absent unless a private collector is configured. - - it: omits the CA env, volume, and mount when no caSecretName is set - template: deployment.yaml - set: - billingMetrics: - enabled: true - asserts: - - notContains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls-ca - secret: - secretName: billing-ca - - notContains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: billing-metrics-mtls-ca - mountPath: /etc/litellm/billing-mtls-ca - readOnly: true - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_CA_CERT - value: /etc/litellm/billing-mtls-ca/ca.crt - - - it: mounts the CA secret when caSecretName is set - template: deployment.yaml - set: - billingMetrics: - enabled: true - caSecretName: billing-ca - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_CA_CERT - value: /etc/litellm/billing-mtls-ca/ca.crt - - contains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls-ca - secret: - secretName: billing-ca - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: billing-metrics-mtls-ca - mountPath: /etc/litellm/billing-mtls-ca - readOnly: true - - - it: passes the export interval through only when set - template: deployment.yaml - set: - billingMetrics: - enabled: true - exportIntervalMs: 5000 - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_EXPORT_INTERVAL_MS - value: "5000" - - - it: omits the export interval when unset - template: deployment.yaml - set: - billingMetrics: - enabled: true - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_EXPORT_INTERVAL_MS - value: "60000" - - # Kubernetes resolves duplicate env names last-wins, so the chart-owned billing - # entries must render after .Values.envVars or a user could silently redirect - # the metering export. The three billing entries are the last ones emitted here - # (migrationJob, which appends DISABLE_SCHEMA_UPDATE, is off for this case). - - it: renders the billing endpoint after envVars so it cannot be shadowed - template: deployment.yaml - set: - migrationJob: - enabled: false - billingMetrics: - enabled: true - envVars: - LITELLM_BILLING_METRICS_ENDPOINT: https://shadowed.example - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://shadowed.example - - equal: - path: spec.template.spec.containers[0].env[-3] - value: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://telemetry.litellm.ai - - equal: - path: spec.template.spec.containers[0].env[-2].name - value: LITELLM_BILLING_METRICS_CLIENT_CERT - - equal: - path: spec.template.spec.containers[0].env[-1].name - value: LITELLM_BILLING_METRICS_CLIENT_KEY - - - it: keeps user-supplied volumes and mounts alongside the billing secret - template: deployment.yaml - set: - billingMetrics: - enabled: true - volumes: - - name: custom-callbacks - configMap: - name: my-callbacks - volumeMounts: - - name: custom-callbacks - mountPath: /app/callbacks - asserts: - - contains: - path: spec.template.spec.volumes - content: - name: custom-callbacks - configMap: - name: my-callbacks - - contains: - path: spec.template.spec.volumes - content: - name: billing-metrics-mtls - secret: - secretName: litellm-billing-metrics-mtls - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: custom-callbacks - mountPath: /app/callbacks - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: billing-metrics-mtls - mountPath: /etc/litellm/billing-mtls - readOnly: true - - - it: still mounts the proxy config when enabled - template: deployment.yaml - set: - billingMetrics: - enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: litellm-config - mountPath: /etc/litellm/config.yaml - subPath: config.yaml - - # Only the proxy serves billable traffic. The migrations Job must never mount - # the client certificate, and it renders its own env and volumes, so nothing - # stops a future edit from wiring the billing include into it by mistake. - - it: does not touch the migrations job when enabled - template: migrations-job.yaml - set: - billingMetrics: - enabled: true - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://telemetry.litellm.ai - - notExists: - path: spec.template.spec.containers[0].volumeMounts - - notExists: - path: spec.template.spec.volumes - - - it: fails loudly when enabled with an emptied secretName - template: deployment.yaml - set: - billingMetrics: - enabled: true - secretName: "" - asserts: - - failedTemplate: - errorMessage: billingMetrics.secretName is required when billingMetrics.enabled is true (an existing Secret with tls.crt and tls.key) - - - it: fails loudly when enabled without an endpoint - template: deployment.yaml - set: - billingMetrics: - enabled: true - endpoint: "" - asserts: - - failedTemplate: - errorMessage: billingMetrics.endpoint is required when billingMetrics.enabled is true diff --git a/helm/litellm-helm/tests/bundled_db_images_tests.yaml b/helm/litellm-helm/tests/bundled_db_images_tests.yaml deleted file mode 100644 index 8f0860c2721..00000000000 --- a/helm/litellm-helm/tests/bundled_db_images_tests.yaml +++ /dev/null @@ -1,94 +0,0 @@ -suite: test bundled database images -templates: - - charts/postgresql/templates/primary/statefulset.yaml - - charts/redis/templates/master/application.yaml - - charts/redis/templates/configmap.yaml - - charts/redis/templates/health-configmap.yaml - - charts/redis/templates/scripts-configmap.yaml - - charts/redis/templates/secret.yaml - - secret-dbcredentials.yaml - - templates/tests/test-servicemonitor.yaml -tests: - - it: should pull the bundled postgres from a repository that still publishes the pinned tag - template: charts/postgresql/templates/primary/statefulset.yaml - set: - db.deployStandalone: true - asserts: - - equal: - path: spec.template.spec.containers[0].image - value: docker.io/bitnamilegacy/postgresql:16.2.0-debian-12-r6 - - - it: should pull the bundled postgres metrics exporter from the same repository - template: charts/postgresql/templates/primary/statefulset.yaml - set: - db.deployStandalone: true - postgresql.metrics.enabled: true - asserts: - - equal: - path: spec.template.spec.containers[1].image - value: docker.io/bitnamilegacy/postgres-exporter:0.15.0-debian-12-r14 - - - it: should run the bundled postgres init container from the same repository - template: charts/postgresql/templates/primary/statefulset.yaml - set: - db.deployStandalone: true - postgresql.volumePermissions.enabled: true - asserts: - - equal: - path: spec.template.spec.initContainers[0].image - value: docker.io/bitnamilegacy/os-shell:12-debian-12-r16 - - - it: should pull the bundled redis from a repository that still publishes the pinned tag - template: charts/redis/templates/master/application.yaml - set: - redis.enabled: true - asserts: - - equal: - path: spec.template.spec.containers[0].image - value: docker.io/bitnamilegacy/redis:7.2.4-debian-12-r9 - - - it: should reject a floating postgres tag that could cross a major version on an existing volume - template: secret-dbcredentials.yaml - set: - db.deployStandalone: true - postgresql.image.tag: latest - asserts: - - failedTemplate: - errorMessage: 'postgresql.image.tag must be pinned to an explicit version when db.deployStandalone is true (got "latest"). An unpinned tag can start a different PostgreSQL major against the existing data directory, which makes the database unreadable and is not recoverable in place. Crossing a major version requires a dump and restore.' - - - it: should reject an empty postgres tag - template: secret-dbcredentials.yaml - set: - db.deployStandalone: true - postgresql.image.tag: "" - asserts: - - failedTemplate: - errorMessage: 'postgresql.image.tag must be pinned to an explicit version when db.deployStandalone is true (got ""). An unpinned tag can start a different PostgreSQL major against the existing data directory, which makes the database unreadable and is not recoverable in place. Crossing a major version requires a dump and restore.' - - - it: should accept an empty postgres tag when the image is pinned by digest - template: secret-dbcredentials.yaml - set: - db.deployStandalone: true - postgresql.image.tag: "" - postgresql.image.digest: sha256:0d0e2f1a5b3c4d6e7f8091a2b3c4d5e6f708192a3b4c5d6e7f8091a2b3c4d5e6 - asserts: - - hasDocuments: - count: 1 - - - it: should run the servicemonitor test pod from a pinned image - template: templates/tests/test-servicemonitor.yaml - set: - serviceMonitor.enabled: true - asserts: - - equal: - path: spec.containers[0].image - value: docker.io/bitnamilegacy/kubectl:1.29.2-debian-12-r3 - - - it: should not constrain the postgres tag when the bundled database is not deployed - template: secret-dbcredentials.yaml - set: - db.deployStandalone: false - postgresql.image.tag: latest - asserts: - - hasDocuments: - count: 0 diff --git a/helm/litellm-helm/tests/collector_tests.yaml b/helm/litellm-helm/tests/collector_tests.yaml deleted file mode 100644 index 0340b1161b7..00000000000 --- a/helm/litellm-helm/tests/collector_tests.yaml +++ /dev/null @@ -1,272 +0,0 @@ -suite: test collector sidecar -templates: - - deployment.yaml - - hpa.yaml - - configmap-litellm.yaml -tests: - - it: should run the proxy alone with no collector env by default - template: deployment.yaml - asserts: - - lengthEqual: - path: spec.template.spec.containers - count: 1 - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ENABLED - value: "true" - - notContains: - path: spec.template.spec.volumes - content: - name: collector-socket - any: true - - - it: should add the sidecar on the same image and point both containers at the unix socket - template: deployment.yaml - set: - image.tag: test - db.connectionPool.enabled: true - collector.enabled: true - collector.resources: - requests: - cpu: 500m - memory: 1Gi - limits: - cpu: "1" - memory: 2Gi - asserts: - - lengthEqual: - path: spec.template.spec.containers - count: 2 - - equal: - path: spec.template.spec.containers[1].name - value: litellm-collector - - equal: - path: spec.template.spec.containers[1].image - value: ghcr.io/berriai/litellm:test - - equal: - path: spec.template.spec.containers[1].command - value: [python, -m, litellm.proxy.collector] - - equal: - path: spec.template.spec.containers[1].resources.requests.cpu - value: 500m - - equal: - path: spec.template.spec.containers[1].resources.limits.memory - value: 2Gi - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ENABLED - value: "true" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ADDRESS - value: unix:///var/run/litellm/collector.sock - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_BUFFER_SIZE - value: "1000" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ON_UNAVAILABLE - value: fallback - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_JOB_ROLE - value: collector - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_JOB_ROLE - value: collector - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_COLLECTOR_ADDRESS - value: unix:///var/run/litellm/collector.sock - - contains: - path: spec.template.spec.containers[1].env - content: - name: CONFIG_FILE_PATH - value: /etc/litellm/config.yaml - - contains: - path: spec.template.spec.containers[1].env - content: - name: DATABASE_HOST - value: RELEASE-NAME-postgresql - - contains: - path: spec.template.spec.containers[1].env - content: - name: DATABASE_PASSWORD - valueFrom: - secretKeyRef: - name: RELEASE-NAME-litellm-dbcredentials - key: password - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_ENABLED - value: "true" - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: "20" - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: collector-socket - mountPath: /var/run/litellm - - contains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: collector-socket - mountPath: /var/run/litellm - - contains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: litellm-config - mountPath: /etc/litellm/config.yaml - subPath: config.yaml - - contains: - path: spec.template.spec.volumes - content: - name: collector-socket - emptyDir: - sizeLimit: 1Mi - - - it: should skip the socket volume and pass the policy through on tcp transport - template: deployment.yaml - set: - collector.enabled: true - collector.address: tcp://127.0.0.1:4100 - collector.onUnavailable: drop - collector.bufferSize: 50 - envVars: - CONFIG_FILE_PATH: /custom/config.yaml - asserts: - - lengthEqual: - path: spec.template.spec.containers - count: 2 - - notContains: - path: spec.template.spec.containers[1].env - content: - name: CONFIG_FILE_PATH - value: /etc/litellm/config.yaml - - contains: - path: spec.template.spec.containers[1].env - content: - name: CONFIG_FILE_PATH - value: /custom/config.yaml - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ADDRESS - value: tcp://127.0.0.1:4100 - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_ON_UNAVAILABLE - value: drop - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_COLLECTOR_BUFFER_SIZE - value: "50" - - notContains: - path: spec.template.spec.volumes - content: - name: collector-socket - any: true - - - it: should keep metrics and billing env on the proxy container only - template: deployment.yaml - set: - collector.enabled: true - metricsServer.enabled: true - metricsServer.port: 9090 - billingMetrics.enabled: true - billingMetrics.endpoint: https://metering.example.com - billingMetrics.secretName: billing-mtls - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: PROMETHEUS_METRICS_PORT - value: "9090" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - value: https://metering.example.com - - notContains: - path: spec.template.spec.containers[1].env - content: - name: PROMETHEUS_METRICS_PORT - any: true - - notContains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_BILLING_METRICS_ENDPOINT - any: true - - notContains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: billing-metrics-mtls - any: true - - - it: should give the sidecar the same scratch mounts as the proxy on a read-only root - template: deployment.yaml - set: - collector.enabled: true - securityContext.readOnlyRootFilesystem: true - asserts: - - contains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: npm - mountPath: /.npm - - contains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: cache - mountPath: /.cache - - contains: - path: spec.template.spec.containers[1].volumeMounts - content: - name: tmp - mountPath: /tmp - - - it: should keep the pod-wide cpu metric unless asked to scale on the proxy container - template: hpa.yaml - set: - autoscaling.enabled: true - collector.enabled: true - asserts: - - equal: { path: "spec.metrics[0].type", value: Resource } - - equal: { path: "spec.metrics[0].resource.name", value: cpu } - - - it: should scale on the proxy container's cpu only when opted in - template: hpa.yaml - set: - autoscaling.enabled: true - collector.enabled: true - collector.scaleOnProxyContainerCpu: true - asserts: - - equal: { path: "spec.metrics[0].type", value: ContainerResource } - - equal: { path: "spec.metrics[0].containerResource.name", value: cpu } - - equal: { path: "spec.metrics[0].containerResource.container", value: litellm } - - equal: { path: "spec.metrics[0].containerResource.target.averageUtilization", value: 60 } - - isNull: { path: "spec.metrics[0].resource" } - - - it: should not switch to the container metric while the sidecar is off - template: hpa.yaml - set: - autoscaling.enabled: true - collector.scaleOnProxyContainerCpu: true - asserts: - - equal: { path: "spec.metrics[0].type", value: Resource } diff --git a/helm/litellm-helm/tests/connection_pool_tests.yaml b/helm/litellm-helm/tests/connection_pool_tests.yaml deleted file mode 100644 index 203082f27ba..00000000000 --- a/helm/litellm-helm/tests/connection_pool_tests.yaml +++ /dev/null @@ -1,112 +0,0 @@ -suite: test in-container connection pool -templates: - - deployment.yaml - - configmap-litellm.yaml -tests: - - it: should not emit pgbouncer env vars by default - template: deployment.yaml - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_ENABLED - value: "true" - - notContains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: "20" - - - it: should enable the pool with the default sizing when connectionPool.enabled is set - template: deployment.yaml - set: - db.connectionPool.enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_ENABLED - value: "true" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: "20" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_MAX_CLIENT_CONN - value: "1000" - - - it: should pass custom sizing through as strings next to the worker count - template: deployment.yaml - set: - numWorkers: 4 - db.connectionPool.enabled: true - db.connectionPool.maxDbConnections: 8 - db.connectionPool.maxClientConn: 400 - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: "8" - - contains: - path: spec.template.spec.containers[0].env - content: - name: LITELLM_PGBOUNCER_MAX_CLIENT_CONN - value: "400" - - contains: - path: spec.template.spec.containers[0].args - content: "4" - - - it: should give the collector sidecar the same pool env as the proxy container - template: deployment.yaml - set: - collector.enabled: true - db.connectionPool.enabled: true - db.connectionPool.maxDbConnections: 8 - db.connectionPool.maxClientConn: 400 - asserts: - - equal: - path: spec.template.spec.containers[1].name - value: litellm-collector - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_ENABLED - value: "true" - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - value: "8" - - contains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_MAX_CLIENT_CONN - value: "400" - - - it: should give the collector sidecar no pool env when the pool is off - template: deployment.yaml - set: - collector.enabled: true - asserts: - - equal: - path: spec.template.spec.containers[1].name - value: litellm-collector - - notContains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_ENABLED - any: true - - notContains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_MAX_DB_CONNECTIONS - any: true - - notContains: - path: spec.template.spec.containers[1].env - content: - name: LITELLM_PGBOUNCER_MAX_CLIENT_CONN - any: true diff --git a/helm/litellm-helm/tests/coordination_redis_tests.yaml b/helm/litellm-helm/tests/coordination_redis_tests.yaml deleted file mode 100644 index 0b58b1e6bc8..00000000000 --- a/helm/litellm-helm/tests/coordination_redis_tests.yaml +++ /dev/null @@ -1,143 +0,0 @@ -suite: test coordination redis -templates: - - configmap-litellm.yaml - - deployment.yaml -tests: - - it: should not render coordination_redis when redis is disabled - template: configmap-litellm.yaml - set: - redis.enabled: false - asserts: - - notMatchRegex: - path: data["config.yaml"] - pattern: coordination_redis - - - it: should not emit redis env vars when redis is disabled - template: deployment.yaml - set: - redis.enabled: false - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_HOST - value: RELEASE-NAME-redis-master - any: true - - - it: should render coordination_redis pointing at the bundled redis when enabled - template: configmap-litellm.yaml - set: - redis.enabled: true - asserts: - - matchRegex: - path: data["config.yaml"] - pattern: "coordination_redis:\n host: os.environ/REDIS_HOST\n password: os.environ/REDIS_PASSWORD\n port: os.environ/REDIS_PORT\n" - - matchRegex: - path: data["config.yaml"] - pattern: "master_key: os.environ/PROXY_MASTER_KEY" - - - it: should emit redis env vars backing the coordination_redis os.environ refs - template: deployment.yaml - set: - redis.enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_HOST - value: RELEASE-NAME-redis-master - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_PORT - value: "6379" - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_PASSWORD - valueFrom: - secretKeyRef: - name: RELEASE-NAME-redis - key: redis-password - - - it: should not render coordination_redis when coordination is opted out - template: configmap-litellm.yaml - set: - redis.enabled: true - redis.coordination.enabled: false - asserts: - - notMatchRegex: - path: data["config.yaml"] - pattern: coordination_redis - - - it: should keep emitting redis env vars when coordination is opted out - template: deployment.yaml - set: - redis.enabled: true - redis.coordination.enabled: false - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_HOST - value: RELEASE-NAME-redis-master - - - it: should not clobber a user supplied coordination_redis block - template: configmap-litellm.yaml - set: - redis.enabled: true - proxy_config.general_settings.coordination_redis: - url: os.environ/COORDINATION_REDIS_URL - asserts: - - matchRegex: - path: data["config.yaml"] - pattern: "coordination_redis:\n url: os.environ/COORDINATION_REDIS_URL\n" - - notMatchRegex: - path: data["config.yaml"] - pattern: "host: os.environ/REDIS_HOST" - - - it: should render sentinel_nodes and service_name in sentinel mode - template: configmap-litellm.yaml - set: - redis.enabled: true - redis.architecture: replication - redis.sentinel.enabled: true - asserts: - # The sentinel Service the redis subchart renders is "-redis", and a - # plain client cannot speak the sentinel protocol, so host/port must not appear - - matchRegex: - path: data["config.yaml"] - pattern: "coordination_redis:\n password: os.environ/REDIS_PASSWORD\n sentinel_nodes:\n - - RELEASE-NAME-redis\n - 26379\n service_name: mymaster\n" - - notMatchRegex: - path: data["config.yaml"] - pattern: "host: os.environ/REDIS_HOST" - - - it: should carry a custom sentinel masterSet into service_name - template: configmap-litellm.yaml - set: - redis.enabled: true - redis.architecture: replication - redis.sentinel.enabled: true - redis.sentinel.masterSet: litellm-master - asserts: - - matchRegex: - path: data["config.yaml"] - pattern: "service_name: litellm-master" - - - it: should point REDIS_HOST at the sentinel service in sentinel mode - template: deployment.yaml - set: - redis.enabled: true - redis.architecture: replication - redis.sentinel.enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_HOST - value: RELEASE-NAME-redis - - contains: - path: spec.template.spec.containers[0].env - content: - name: REDIS_PORT - value: "26379" diff --git a/helm/litellm-helm/tests/deployment_command_args_labels_tests.yaml b/helm/litellm-helm/tests/deployment_command_args_labels_tests.yaml deleted file mode 100644 index 6b0d45ebf48..00000000000 --- a/helm/litellm-helm/tests/deployment_command_args_labels_tests.yaml +++ /dev/null @@ -1,68 +0,0 @@ -suite: test deployment command, args, and deploymentLabels -templates: - - deployment.yaml - - configmap-litellm.yaml -tests: - - it: should override args when custom args specified - template: deployment.yaml - set: - args: - - --custom-arg1 - - value1 - - --custom-arg2 - asserts: - - equal: - path: spec.template.spec.containers[0].args - value: - - --custom-arg1 - - value1 - - --custom-arg2 - - it: should set custom command when specified - template: deployment.yaml - set: - command: - - /bin/sh - - -c - asserts: - - equal: - path: spec.template.spec.containers[0].command - value: - - /bin/sh - - -c - - it: should set custom command and args together - template: deployment.yaml - set: - command: - - python - - -u - args: - - my_script.py - - --verbose - asserts: - - equal: - path: spec.template.spec.containers[0].command - value: - - python - - -u - - equal: - path: spec.template.spec.containers[0].args - value: - - my_script.py - - --verbose - - it: should add deploymentLabels to deployment metadata - template: deployment.yaml - set: - deploymentLabels: - environment: production - team: platform - version: v1.2.3 - asserts: - - equal: - path: metadata.labels.environment - value: production - - equal: - path: metadata.labels.team - value: platform - - equal: - path: metadata.labels.version - value: v1.2.3 diff --git a/helm/litellm-helm/tests/deployment_tests.yaml b/helm/litellm-helm/tests/deployment_tests.yaml deleted file mode 100644 index ee946038202..00000000000 --- a/helm/litellm-helm/tests/deployment_tests.yaml +++ /dev/null @@ -1,494 +0,0 @@ -suite: test deployment -templates: - - deployment.yaml - - configmap-litellm.yaml -tests: - - it: should work - template: deployment.yaml - set: - image.tag: test - asserts: - - isKind: - of: Deployment - - matchRegex: - path: metadata.name - pattern: -litellm$ - - equal: - path: spec.template.spec.containers[0].image - value: ghcr.io/berriai/litellm:test - - it: should work with tolerations - template: deployment.yaml - set: - tolerations: - - key: node-role.kubernetes.io/master - operator: Exists - effect: NoSchedule - asserts: - - equal: - path: spec.template.spec.tolerations[0].key - value: node-role.kubernetes.io/master - - equal: - path: spec.template.spec.tolerations[0].operator - value: Exists - - it: should work with affinity - template: deployment.yaml - set: - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - antarctica-east1 - asserts: - - equal: - path: spec.template.spec.affinity.nodeAffinity.requiredDuringSchedulingIgnoredDuringExecution.nodeSelectorTerms[0].matchExpressions[0].key - value: topology.kubernetes.io/zone - - equal: - path: spec.template.spec.affinity.nodeAffinity.requiredDuringSchedulingIgnoredDuringExecution.nodeSelectorTerms[0].matchExpressions[0].operator - value: In - - equal: - path: spec.template.spec.affinity.nodeAffinity.requiredDuringSchedulingIgnoredDuringExecution.nodeSelectorTerms[0].matchExpressions[0].values[0] - value: antarctica-east1 - - it: should work without masterkeySecretName or masterkeySecretKey - template: deployment.yaml - set: - masterkeySecretName: "" - masterkeySecretKey: "" - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: PROXY_MASTER_KEY - valueFrom: - secretKeyRef: - name: RELEASE-NAME-litellm-masterkey - key: masterkey - - it: should work with masterkeySecretName and masterkeySecretKey - template: deployment.yaml - set: - masterkeySecretName: my-secret - masterkeySecretKey: my-key - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: PROXY_MASTER_KEY - valueFrom: - secretKeyRef: - name: my-secret - key: my-key - - it: should inject DATABASE_READER_HOST from readReplicaEndpointKey before DATABASE_URL_READ_REPLICA - template: deployment.yaml - set: - db: - deployStandalone: false - useExisting: true - secret: - name: postgres - usernameKey: username - passwordKey: password - readReplicaEndpointKey: reader-host - readReplicaUrl: postgresql://$(DATABASE_USERNAME):$(DATABASE_PASSWORD)@$(DATABASE_READER_HOST):5432/$(DATABASE_NAME)?sslmode=require - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_READER_HOST - valueFrom: - secretKeyRef: - name: postgres - key: reader-host - - contains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_URL_READ_REPLICA - value: postgresql://$(DATABASE_USERNAME):$(DATABASE_PASSWORD)@$(DATABASE_READER_HOST):5432/$(DATABASE_NAME)?sslmode=require - # $(VAR) interpolation only resolves vars defined EARLIER in the env - # array, so the reader host must precede the composed URL - - equal: - path: spec.template.spec.containers[0].env[7].name - value: DATABASE_READER_HOST - - equal: - path: spec.template.spec.containers[0].env[8].name - value: DATABASE_URL_READ_REPLICA - - it: should omit reader host when readReplicaUrl is unset - template: deployment.yaml - set: - db: - deployStandalone: false - useExisting: true - secret: - name: postgres - usernameKey: username - passwordKey: password - readReplicaEndpointKey: reader-host - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_READER_HOST - valueFrom: - secretKeyRef: - name: postgres - key: reader-host - - it: should prefer readReplicaUrlKey over readReplicaEndpointKey composition - template: deployment.yaml - set: - db: - useExisting: true - secret: - name: postgres - usernameKey: username - passwordKey: password - readReplicaUrlKey: reader-url - readReplicaEndpointKey: reader-host - readReplicaUrl: postgresql://ignored - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_URL_READ_REPLICA - valueFrom: - secretKeyRef: - name: postgres - key: reader-url - - notContains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_URL_READ_REPLICA - value: postgresql://ignored - # the unused reader-host secret ref must be suppressed so a missing - # key can't fail pod creation - - notContains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_READER_HOST - valueFrom: - secretKeyRef: - name: postgres - key: reader-host - - it: should work with extraEnvVars - template: deployment.yaml - set: - extraEnvVars: - - name: EXTRA_ENV_VAR - valueFrom: - fieldRef: - fieldPath: metadata.labels['env'] - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: EXTRA_ENV_VAR - valueFrom: - fieldRef: - fieldPath: metadata.labels['env'] - - it: should work with both extraEnvVars and envVars - template: deployment.yaml - set: - envVars: - ENV_VAR: ENV_VAR_VALUE - extraEnvVars: - - name: EXTRA_ENV_VAR - value: EXTRA_ENV_VAR_VALUE - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: ENV_VAR - value: ENV_VAR_VALUE - - contains: - path: spec.template.spec.containers[0].env - content: - name: EXTRA_ENV_VAR - value: EXTRA_ENV_VAR_VALUE - - it: should mount existing configmap when create=false - template: deployment.yaml - set: - proxyConfigMap: - create: false - name: my-litellm-config - key: custom.yaml - asserts: - - contains: - path: spec.template.spec.volumes - content: - name: litellm-config - configMap: - name: my-litellm-config - items: - - key: custom.yaml - path: config.yaml - - contains: - path: spec.template.spec.containers[0].volumeMounts - content: - name: litellm-config - mountPath: /etc/litellm/config.yaml - subPath: config.yaml - - it: should work with lifecycle hooks - template: deployment.yaml - set: - lifecycle: - preStop: - exec: - command: - - /bin/sh - - -c - - echo "Container stopping" - asserts: - - exists: - path: spec.template.spec.containers[0].lifecycle - - equal: - path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[0] - value: /bin/sh - - equal: - path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[1] - value: -c - - equal: - path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[2] - value: echo "Container stopping" - - it: should render background health check settings from proxy_config.general_settings - template: configmap-litellm.yaml - set: - proxy_config.general_settings.background_health_checks: true - proxy_config.general_settings.health_check_interval: 240 - proxy_config.general_settings.health_check_concurrency: 16 - proxy_config.general_settings.health_check_details: false - asserts: - - matchRegex: - path: data["config.yaml"] - pattern: '(?m)^\s*background_health_checks:\s*true$' - - matchRegex: - path: data["config.yaml"] - pattern: '(?m)^\s*health_check_interval:\s*240$' - - matchRegex: - path: data["config.yaml"] - pattern: '(?m)^\s*health_check_concurrency:\s*16$' - - matchRegex: - path: data["config.yaml"] - pattern: '(?m)^\s*health_check_details:\s*false$' - - it: should allow overriding liveness, readiness, and startup probes - template: deployment.yaml - set: - livenessProbe: - path: /custom/livez - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 - readinessProbe: - path: /custom/readyz - initialDelaySeconds: 10 - periodSeconds: 20 - timeoutSeconds: 6 - successThreshold: 1 - failureThreshold: 6 - startupProbe: - path: /custom/startupz - initialDelaySeconds: 15 - periodSeconds: 25 - timeoutSeconds: 7 - successThreshold: 1 - failureThreshold: 40 - asserts: - - equal: - path: spec.template.spec.containers[0].livenessProbe.httpGet.path - value: /custom/livez - - equal: - path: spec.template.spec.containers[0].livenessProbe.timeoutSeconds - value: 5 - - equal: - path: spec.template.spec.containers[0].readinessProbe.httpGet.path - value: /custom/readyz - - equal: - path: spec.template.spec.containers[0].readinessProbe.timeoutSeconds - value: 6 - - equal: - path: spec.template.spec.containers[0].startupProbe.httpGet.path - value: /custom/startupz - - equal: - path: spec.template.spec.containers[0].startupProbe.failureThreshold - value: 40 - - it: should render container resources from values - template: deployment.yaml - set: - resources: - limits: - cpu: 500m - memory: 2Gi - requests: - cpu: 250m - memory: 1Gi - asserts: - - equal: - path: spec.template.spec.containers[0].resources.limits.cpu - value: 500m - - equal: - path: spec.template.spec.containers[0].resources.limits.memory - value: 2Gi - - equal: - path: spec.template.spec.containers[0].resources.requests.cpu - value: 250m - - equal: - path: spec.template.spec.containers[0].resources.requests.memory - value: 1Gi - - it: should keep default probes and empty resources unchanged - template: deployment.yaml - asserts: - - equal: - path: spec.template.spec.containers[0].livenessProbe.httpGet.path - value: /health/liveliness - - equal: - path: spec.template.spec.containers[0].livenessProbe.initialDelaySeconds - value: 0 - - equal: - path: spec.template.spec.containers[0].livenessProbe.periodSeconds - value: 15 - - equal: - path: spec.template.spec.containers[0].livenessProbe.timeoutSeconds - value: 5 - - equal: - path: spec.template.spec.containers[0].livenessProbe.successThreshold - value: 1 - - equal: - path: spec.template.spec.containers[0].livenessProbe.failureThreshold - value: 5 - - equal: - path: spec.template.spec.containers[0].readinessProbe.httpGet.path - value: /health/readiness - - equal: - path: spec.template.spec.containers[0].readinessProbe.initialDelaySeconds - value: 0 - - equal: - path: spec.template.spec.containers[0].readinessProbe.periodSeconds - value: 10 - - equal: - path: spec.template.spec.containers[0].readinessProbe.timeoutSeconds - value: 5 - - equal: - path: spec.template.spec.containers[0].readinessProbe.successThreshold - value: 1 - - equal: - path: spec.template.spec.containers[0].readinessProbe.failureThreshold - value: 3 - - equal: - path: spec.template.spec.containers[0].startupProbe.httpGet.path - value: /health/readiness - - equal: - path: spec.template.spec.containers[0].startupProbe.initialDelaySeconds - value: 0 - - equal: - path: spec.template.spec.containers[0].startupProbe.periodSeconds - value: 10 - - equal: - path: spec.template.spec.containers[0].startupProbe.timeoutSeconds - value: 5 - - equal: - path: spec.template.spec.containers[0].startupProbe.successThreshold - value: 1 - - equal: - path: spec.template.spec.containers[0].startupProbe.failureThreshold - value: 30 - - equal: - path: spec.template.spec.containers[0].resources - value: {} - - it: should be able to set minReadySeconds - template: deployment.yaml - set: - deploymentMinReadySeconds: 5 - asserts: - - equal: - path: spec.minReadySeconds - value: 5 - - it: should have minReadySeconds absent when deploymentMinReadySeconds is not set - template: deployment.yaml - asserts: - - notExists: - path: spec.minReadySeconds - - it: should work with extraInitContainers - template: deployment.yaml - set: - extraInitContainers: - - name: init-test - image: busybox:latest - command: ["echo", "hello"] - asserts: - - contains: - path: spec.template.spec.initContainers - content: - name: init-test - image: busybox:latest - command: ["echo", "hello"] - - it: should support tpl in extraInitContainers - template: deployment.yaml - set: - image: - repository: ghcr.io/berriai/litellm - tag: test - extraInitContainers: - - name: init-tpl - image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}" - command: ["echo", "hello"] - asserts: - - contains: - path: spec.template.spec.initContainers - content: - name: init-tpl - image: "ghcr.io/berriai/litellm:test" - command: ["echo", "hello"] - - it: should work with extraContainers - template: deployment.yaml - set: - extraContainers: - - name: sidecar - image: busybox:latest - asserts: - - contains: - path: spec.template.spec.containers - content: - name: sidecar - image: busybox:latest - - it: should support tpl in extraContainers - template: deployment.yaml - set: - image: - repository: ghcr.io/berriai/litellm - tag: test - extraContainers: - - name: sidecar-tpl - image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}" - asserts: - - contains: - path: spec.template.spec.containers - content: - name: sidecar-tpl - image: "ghcr.io/berriai/litellm:test" - - it: should support tpl in podAnnotations - template: deployment.yaml - set: - image: - repository: ghcr.io/berriai/litellm - tag: test - # Mirrors the real-world scenario this feature unblocks: - # user disables the built-in ConfigMap (and its built-in checksum/config - # annotation) and re-implements checksum/config themselves via tpl. - proxyConfigMap: - create: false - podAnnotations: - checksum/config: "{{ .Values.image.tag }}" - example.com/some-key: "{{ .Values.image.repository }}" - example.com/literal: "plain-string-value" - asserts: - - equal: - path: spec.template.metadata.annotations["checksum/config"] - value: "test" - - equal: - path: spec.template.metadata.annotations["example.com/some-key"] - value: "ghcr.io/berriai/litellm" - - equal: - path: spec.template.metadata.annotations["example.com/literal"] - value: "plain-string-value" diff --git a/helm/litellm-helm/tests/hpa_tests.yaml b/helm/litellm-helm/tests/hpa_tests.yaml deleted file mode 100644 index e446f58c8fe..00000000000 --- a/helm/litellm-helm/tests/hpa_tests.yaml +++ /dev/null @@ -1,144 +0,0 @@ -suite: "hpa" -templates: - - hpa.yaml -tests: - - it: "renders behavior when set" - set: - autoscaling.enabled: true - autoscaling.behavior: - scaleUp: - stabilizationWindowSeconds: 60 - policies: - - type: Pods - value: 2 - periodSeconds: 60 - scaleDown: - stabilizationWindowSeconds: 90 - policies: - - type: Pods - value: 1 - periodSeconds: 60 - asserts: - - isKind: { of: HorizontalPodAutoscaler } - - equal: { path: spec.behavior.scaleUp.stabilizationWindowSeconds, value: 60 } - - equal: { path: spec.behavior.scaleDown.stabilizationWindowSeconds, value: 90 } - - - it: "does not render behavior when not set" - set: - autoscaling.enabled: true - asserts: - - isKind: { of: HorizontalPodAutoscaler } - - isNull: { path: spec.behavior } - - - it: "scales on cpu at the documented 60 percent by default" - set: - autoscaling.enabled: true - asserts: - - isKind: { of: HorizontalPodAutoscaler } - - equal: { path: "spec.metrics[0].resource.name", value: cpu } - - equal: { path: "spec.metrics[0].resource.target.type", value: Utilization } - - equal: { path: "spec.metrics[0].resource.target.averageUtilization", value: 60 } - - - it: "does not scale on memory by default" - set: - autoscaling.enabled: true - asserts: - - lengthEqual: { path: spec.metrics, count: 1 } - - - it: "honours an explicit cpu target override" - set: - autoscaling.enabled: true - autoscaling.targetCPUUtilizationPercentage: 75 - asserts: - - equal: { path: "spec.metrics[0].resource.target.averageUtilization", value: 75 } - - - it: "renders a memory metric only when a memory target is set" - set: - autoscaling.enabled: true - autoscaling.targetMemoryUtilizationPercentage: 80 - asserts: - - lengthEqual: { path: spec.metrics, count: 2 } - - equal: { path: "spec.metrics[1].resource.name", value: memory } - - equal: { path: "spec.metrics[1].resource.target.averageUtilization", value: 80 } - - - it: "renders no workload metrics by default" - set: - autoscaling.enabled: true - autoscaling.targetMemoryUtilizationPercentage: 80 - asserts: - - lengthEqual: { path: spec.metrics, count: 2 } - - notContains: { path: spec.metrics, content: { type: Pods }, any: true } - - - it: "adds a requests-per-second Pods metric after the cpu metric" - set: - autoscaling.enabled: true - autoscaling.targetRequestsPerSecond: 90 - asserts: - - lengthEqual: { path: spec.metrics, count: 2 } - - equal: { path: "spec.metrics[0].resource.name", value: cpu } - - equal: - path: "spec.metrics[1]" - value: - type: Pods - pods: - metric: { name: litellm_requests_per_second } - target: { type: AverageValue, averageValue: "90" } - - - it: "adds a tokens-per-second Pods metric on its own" - set: - autoscaling.enabled: true - autoscaling.targetTokensPerSecond: 6M - asserts: - - lengthEqual: { path: spec.metrics, count: 2 } - - equal: - path: "spec.metrics[1]" - value: - type: Pods - pods: - metric: { name: litellm_tokens_per_second } - target: { type: AverageValue, averageValue: "6M" } - - notContains: - path: spec.metrics - content: { type: Pods, pods: { metric: { name: litellm_requests_per_second } } } - any: true - - - it: "renders requests, tokens, cpu and memory metrics together" - set: - autoscaling.enabled: true - autoscaling.targetMemoryUtilizationPercentage: 80 - autoscaling.targetRequestsPerSecond: 90 - autoscaling.targetTokensPerSecond: 6000000 - asserts: - - lengthEqual: { path: spec.metrics, count: 4 } - - equal: { path: "spec.metrics[0].resource.name", value: cpu } - - equal: { path: "spec.metrics[1].resource.name", value: memory } - - equal: { path: "spec.metrics[2].pods.metric.name", value: litellm_requests_per_second } - - equal: { path: "spec.metrics[2].pods.target.averageValue", value: "90" } - - equal: { path: "spec.metrics[3].pods.metric.name", value: litellm_tokens_per_second } - - equal: { path: "spec.metrics[3].pods.target.averageValue", value: "6000000" } - - - it: "scales on workload metrics alone when the cpu target is cleared" - set: - autoscaling.enabled: true - autoscaling.targetCPUUtilizationPercentage: null - autoscaling.targetRequestsPerSecond: 90 - autoscaling.targetTokensPerSecond: 6000000 - asserts: - - lengthEqual: { path: spec.metrics, count: 2 } - - notContains: { path: spec.metrics, content: { type: Resource }, any: true } - - equal: { path: "spec.metrics[0].pods.metric.name", value: litellm_requests_per_second } - - equal: { path: "spec.metrics[1].pods.metric.name", value: litellm_tokens_per_second } - - notMatchRegexRaw: { pattern: per_minute } - - - it: "ignores the per-minute keys, which the chart never shipped" - set: - autoscaling.enabled: true - autoscaling.targetRequestsPerMinute: 5400 - autoscaling.targetTokensPerMinute: 360000000 - asserts: - - lengthEqual: { path: spec.metrics, count: 1 } - - notContains: { path: spec.metrics, content: { type: Pods }, any: true } - - - it: "renders no hpa when autoscaling is disabled" - asserts: - - hasDocuments: { count: 0 } diff --git a/helm/litellm-helm/tests/ingress_tests.yaml b/helm/litellm-helm/tests/ingress_tests.yaml deleted file mode 100644 index aad6ecfcee8..00000000000 --- a/helm/litellm-helm/tests/ingress_tests.yaml +++ /dev/null @@ -1,45 +0,0 @@ -suite: Ingress Configuration Tests -templates: - - ingress.yaml -tests: - - it: should not create Ingress by default - asserts: - - hasDocuments: - count: 0 - - - it: should create Ingress when enabled - set: - ingress.enabled: true - asserts: - - hasDocuments: - count: 1 - - isKind: - of: Ingress - - - it: should add custom labels - set: - ingress.enabled: true - ingress.labels: - custom-label: "true" - another-label: "value" - asserts: - - isKind: - of: Ingress - - equal: - path: metadata.labels.custom-label - value: "true" - - equal: - path: metadata.labels.another-label - value: "value" - - - it: should add annotations - set: - ingress.enabled: true - ingress.annotations: - kubernetes.io/ingress.class: "nginx" - asserts: - - isKind: - of: Ingress - - equal: - path: metadata.annotations["kubernetes.io/ingress.class"] - value: "nginx" diff --git a/helm/litellm-helm/tests/keda_tests.yaml b/helm/litellm-helm/tests/keda_tests.yaml deleted file mode 100644 index c9598646223..00000000000 --- a/helm/litellm-helm/tests/keda_tests.yaml +++ /dev/null @@ -1,106 +0,0 @@ -suite: "keda" -templates: - - keda.yaml -release: - name: rel - namespace: llm -tests: - - it: "renders no scaled object by default" - asserts: - - hasDocuments: { count: 0 } - - - it: "passes user triggers through and adds no prometheus triggers by default" - set: - keda.enabled: true - keda.triggers: - - type: cpu - metricType: Utilization - metadata: { value: "60" } - asserts: - - isKind: { of: ScaledObject } - - equal: - path: spec.triggers - value: - - type: cpu - metricType: Utilization - metadata: { value: "60" } - - - it: "scales on release-wide requests per second divided by the per-replica target" - set: - keda.enabled: true - keda.prometheus.serverAddress: http://prometheus-operated.monitoring.svc:9090 - keda.prometheus.requestsPerSecond: 90 - asserts: - - lengthEqual: { path: spec.triggers, count: 1 } - - equal: - path: "spec.triggers[0]" - value: - type: prometheus - metadata: - serverAddress: http://prometheus-operated.monitoring.svc:9090 - threshold: "90" - query: sum(rate(litellm_proxy_total_requests_metric_total{namespace="llm",job="rel-litellm"}[1m])) - - - it: "scales on tokens per second on its own" - set: - keda.enabled: true - keda.prometheus.serverAddress: http://prom:9090 - keda.prometheus.tokensPerSecond: 6000000 - asserts: - - lengthEqual: { path: spec.triggers, count: 1 } - - equal: { path: "spec.triggers[0].type", value: prometheus } - - equal: { path: "spec.triggers[0].metadata.threshold", value: "6000000" } - - equal: - path: "spec.triggers[0].metadata.query" - value: sum(rate(litellm_total_tokens_metric_total{namespace="llm",job="rel-litellm"}[1m])) - - - it: "appends requests and tokens triggers after user triggers and selects the metrics service job" - set: - keda.enabled: true - metricsServer.enabled: true - keda.triggers: - - type: cpu - metricType: Utilization - metadata: { value: "60" } - keda.prometheus.serverAddress: http://prom:9090 - keda.prometheus.requestsPerSecond: 90 - keda.prometheus.tokensPerSecond: 6000000 - asserts: - - lengthEqual: { path: spec.triggers, count: 3 } - - equal: { path: "spec.triggers[0].type", value: cpu } - - equal: { path: "spec.triggers[1].metadata.threshold", value: "90" } - - equal: - path: "spec.triggers[1].metadata.query" - value: sum(rate(litellm_proxy_total_requests_metric_total{namespace="llm",job="rel-litellm-metrics"}[1m])) - - equal: { path: "spec.triggers[2].metadata.threshold", value: "6000000" } - - equal: - path: "spec.triggers[2].metadata.query" - value: sum(rate(litellm_total_tokens_metric_total{namespace="llm",job="rel-litellm-metrics"}[1m])) - - notMatchRegexRaw: { pattern: "\\* *60|per_minute|PerMinute" } - - - it: "ignores the per-minute keys, which the chart never shipped" - set: - keda.enabled: true - keda.prometheus.serverAddress: http://prom:9090 - keda.prometheus.requestsPerMinute: 5400 - keda.prometheus.tokensPerMinute: 360000000 - asserts: - - isKind: { of: ScaledObject } - - isNullOrEmpty: { path: spec.triggers } - - - it: "refuses a workload target without a prometheus server address" - set: - keda.enabled: true - keda.prometheus.requestsPerSecond: 90 - asserts: - - failedTemplate: - errorMessage: keda.prometheus.serverAddress is required when keda.prometheus.requestsPerSecond or tokensPerSecond is set - - - it: "yields to the hpa when both autoscalers are enabled" - set: - autoscaling.enabled: true - keda.enabled: true - keda.prometheus.serverAddress: http://prom:9090 - keda.prometheus.requestsPerSecond: 90 - asserts: - - hasDocuments: { count: 0 } diff --git a/helm/litellm-helm/tests/masterkey-secret_tests.yaml b/helm/litellm-helm/tests/masterkey-secret_tests.yaml deleted file mode 100644 index 296f26755b8..00000000000 --- a/helm/litellm-helm/tests/masterkey-secret_tests.yaml +++ /dev/null @@ -1,71 +0,0 @@ -suite: test masterkey secret -templates: - - secret-masterkey.yaml -tests: - - it: should create a secret if masterkeySecretName is not set. should start with sk-xxxx (base64 encoded as c2st*) - template: secret-masterkey.yaml - set: - masterkeySecretName: "" - asserts: - - isKind: - of: Secret - - matchRegex: - path: data.masterkey - pattern: ^c2st - # Note: The masterkey is generated as "sk-<18-random-chars>" in plain text, - # but stored as base64 encoded in Kubernetes secret (requirement). - # "sk-" base64 encodes to "c2st", so we check for "^c2st" pattern. - - it: should reuse the master key already stored in the cluster instead of generating a new one on upgrade - template: secret-masterkey.yaml - set: - masterkeySecretName: "" - kubernetesProvider: - scheme: - "v1/Secret": - gvr: - version: "v1" - resource: "secrets" - namespaced: true - objects: - - kind: Secret - apiVersion: v1 - metadata: - name: RELEASE-NAME-litellm-masterkey - namespace: NAMESPACE - data: - masterkey: c2stZXhpc3Rpbmcta2V5 - asserts: - - equal: - path: data.masterkey - value: c2stZXhpc3Rpbmcta2V5 - - it: should let an explicit masterkey value override the one already stored in the cluster - template: secret-masterkey.yaml - set: - masterkeySecretName: "" - masterkey: sk-explicit - kubernetesProvider: - scheme: - "v1/Secret": - gvr: - version: "v1" - resource: "secrets" - namespaced: true - objects: - - kind: Secret - apiVersion: v1 - metadata: - name: RELEASE-NAME-litellm-masterkey - namespace: NAMESPACE - data: - masterkey: c2stZXhpc3Rpbmcta2V5 - asserts: - - equal: - path: data.masterkey - value: c2stZXhwbGljaXQ= - - it: should not create a secret if masterkeySecretName is set - template: secret-masterkey.yaml - set: - masterkeySecretName: my-secret - asserts: - - hasDocuments: - count: 0 diff --git a/helm/litellm-helm/tests/metrics_server_tests.yaml b/helm/litellm-helm/tests/metrics_server_tests.yaml deleted file mode 100644 index 085d69ac640..00000000000 --- a/helm/litellm-helm/tests/metrics_server_tests.yaml +++ /dev/null @@ -1,106 +0,0 @@ -suite: separate metrics server -templates: - - configmap-litellm.yaml - - deployment.yaml - - service.yaml - - service-metrics.yaml - - servicemonitor.yaml -tests: - - it: should not expose a metrics port or PROMETHEUS_METRICS_PORT by default - asserts: - - notContains: - path: spec.template.spec.containers[0].ports - content: - name: metrics - any: true - template: deployment.yaml - - notContains: - path: spec.template.spec.containers[0].env - content: - name: PROMETHEUS_METRICS_PORT - any: true - template: deployment.yaml - - lengthEqual: - path: spec.ports - count: 1 - template: service.yaml - - hasDocuments: - count: 0 - template: service-metrics.yaml - - - it: should scrape the proxy port when the metrics server is disabled - template: servicemonitor.yaml - set: - serviceMonitor.enabled: true - asserts: - - equal: - path: spec.endpoints[0].port - value: http - - - it: should wire the separate metrics server through container, a ClusterIP metrics service and servicemonitor - set: - metricsServer.enabled: true - metricsServer.port: 4101 - serviceMonitor.enabled: true - service.type: LoadBalancer - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: PROMETHEUS_METRICS_PORT - value: "4101" - template: deployment.yaml - - contains: - path: spec.template.spec.containers[0].ports - content: - name: metrics - containerPort: 4101 - protocol: TCP - template: deployment.yaml - - lengthEqual: - path: spec.ports - count: 1 - template: service.yaml - - equal: - path: spec.type - value: LoadBalancer - template: service.yaml - - equal: - path: metadata.name - value: RELEASE-NAME-litellm-metrics - template: service-metrics.yaml - - equal: - path: spec.type - value: ClusterIP - template: service-metrics.yaml - - equal: - path: spec.ports - value: - - port: 4101 - targetPort: metrics - protocol: TCP - name: metrics - template: service-metrics.yaml - - equal: - path: spec.selector - value: - app.kubernetes.io/name: litellm - app.kubernetes.io/instance: RELEASE-NAME - template: service-metrics.yaml - - equal: - path: spec.endpoints[0].port - value: metrics - template: servicemonitor.yaml - - equal: - path: spec.endpoints[0].path - value: /metrics/ - template: servicemonitor.yaml - - - it: should reject a metrics port equal to the proxy port - template: deployment.yaml - set: - metricsServer.enabled: true - metricsServer.port: 4000 - asserts: - - failedTemplate: - errorMessage: metricsServer.port must differ from service.port diff --git a/helm/litellm-helm/tests/migrations-job_tests.yaml b/helm/litellm-helm/tests/migrations-job_tests.yaml deleted file mode 100644 index 1fe545636d4..00000000000 --- a/helm/litellm-helm/tests/migrations-job_tests.yaml +++ /dev/null @@ -1,344 +0,0 @@ -suite: test migrations job -templates: - - migrations-job.yaml -tests: - - it: should work with envVars - template: migrations-job.yaml - set: - envVars: - TEST_ENV_VAR: "test_value" - ANOTHER_VAR: "another_value" - migrationJob: - enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: TEST_ENV_VAR - value: "test_value" - - contains: - path: spec.template.spec.containers[0].env - content: - name: ANOTHER_VAR - value: "another_value" - - - it: should work with extraEnvVars - template: migrations-job.yaml - set: - extraEnvVars: - - name: EXTRA_ENV_VAR - valueFrom: - fieldRef: - fieldPath: metadata.labels['env'] - - name: SIMPLE_EXTRA_VAR - value: "simple_value" - migrationJob: - enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: EXTRA_ENV_VAR - valueFrom: - fieldRef: - fieldPath: metadata.labels['env'] - - contains: - path: spec.template.spec.containers[0].env - content: - name: SIMPLE_EXTRA_VAR - value: "simple_value" - - - it: should work with both envVars and extraEnvVars - template: migrations-job.yaml - set: - envVars: - ENV_VAR: "env_var_value" - extraEnvVars: - - name: EXTRA_ENV_VAR - value: "extra_env_var_value" - migrationJob: - enabled: true - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: ENV_VAR - value: "env_var_value" - - contains: - path: spec.template.spec.containers[0].env - content: - name: EXTRA_ENV_VAR - value: "extra_env_var_value" - - - it: should not render when migrations job is disabled - template: migrations-job.yaml - set: - migrationJob: - enabled: false - asserts: - - hasDocuments: - count: 0 - - - it: should still include default env vars - template: migrations-job.yaml - set: - envVars: - CUSTOM_VAR: "custom_value" - migrationJob: - enabled: true - db: - useExisting: true - endpoint: "test-db" - database: "testdb" - url: "postgresql://user:pass@test-db:5432/testdb" - secret: - name: "test-secret" - usernameKey: "username" - passwordKey: "password" - asserts: - - contains: - path: spec.template.spec.containers[0].env - content: - name: DISABLE_SCHEMA_UPDATE - value: "false" - - contains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_HOST - value: "test-db" - - contains: - path: spec.template.spec.containers[0].env - content: - name: CUSTOM_VAR - value: "custom_value" - - - it: should not include DATABASE_URL when deployStandalone is false - template: migrations-job.yaml - set: - migrationJob: - enabled: true - db: - deployStandalone: false - useExisting: false - asserts: - - notContains: - path: spec.template.spec.containers[0].env - content: - name: DATABASE_URL - - - it: should use default service account for helm hooks when serviceAccount.create is true - template: migrations-job.yaml - set: - migrationJob: - enabled: true - hooks: - helm: - enabled: true - serviceAccount: - create: true - asserts: - - equal: - path: spec.template.spec.serviceAccountName - value: default - - - it: should use migrationJob.serviceAccountName override for helm hooks when serviceAccount.create is true - template: migrations-job.yaml - set: - migrationJob: - enabled: true - serviceAccountName: migration-sa - hooks: - helm: - enabled: true - serviceAccount: - create: true - asserts: - - equal: - path: spec.template.spec.serviceAccountName - value: migration-sa - - - it: should use chart service account when helm hooks are disabled - template: migrations-job.yaml - set: - migrationJob: - enabled: true - hooks: - helm: - enabled: false - serviceAccount: - create: true - name: my-custom-sa - asserts: - - equal: - path: spec.template.spec.serviceAccountName - value: my-custom-sa - - - it: should use pre-existing service account when helm hooks are enabled but serviceAccount.create is false - template: migrations-job.yaml - set: - migrationJob: - enabled: true - hooks: - helm: - enabled: true - serviceAccount: - create: false - name: pre-existing-sa - asserts: - - equal: - path: spec.template.spec.serviceAccountName - value: pre-existing-sa - - it: should work with extraInitContainers - template: migrations-job.yaml - set: - migrationJob: - enabled: true - extraInitContainers: - - name: init-test - image: busybox:latest - command: ["echo", "hello"] - asserts: - - contains: - path: spec.template.spec.initContainers - content: - name: init-test - image: busybox:latest - command: ["echo", "hello"] - - it: should support tpl in extraInitContainers - template: migrations-job.yaml - set: - image: - repository: ghcr.io/berriai/litellm - tag: test - migrationJob: - enabled: true - extraInitContainers: - - name: init-tpl - image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}" - command: ["echo", "hello"] - asserts: - - contains: - path: spec.template.spec.initContainers - content: - name: init-tpl - image: "ghcr.io/berriai/litellm:test" - command: ["echo", "hello"] - - it: should work with extraContainers - template: migrations-job.yaml - set: - migrationJob: - enabled: true - extraContainers: - - name: sidecar - image: busybox:latest - asserts: - - contains: - path: spec.template.spec.containers - content: - name: sidecar - image: busybox:latest - - it: should support tpl in extraContainers - template: migrations-job.yaml - set: - image: - repository: ghcr.io/berriai/litellm - tag: test - migrationJob: - enabled: true - extraContainers: - - name: sidecar-tpl - image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}" - asserts: - - contains: - path: spec.template.spec.containers - content: - name: sidecar-tpl - image: "ghcr.io/berriai/litellm:test" - - it: should render the pod-level securityContext from podSecurityContext - template: migrations-job.yaml - set: - migrationJob: - enabled: true - podSecurityContext: - fsGroup: 10000 - runAsUser: 10000 - runAsNonRoot: true - asserts: - - equal: - path: spec.template.spec.securityContext - value: - fsGroup: 10000 - runAsUser: 10000 - runAsNonRoot: true - - it: should keep the pod-level and container-level securityContext separate - template: migrations-job.yaml - set: - migrationJob: - enabled: true - podSecurityContext: - fsGroup: 10000 - securityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - asserts: - - equal: - path: spec.template.spec.securityContext - value: - fsGroup: 10000 - - equal: - path: spec.template.spec.containers[0].securityContext - value: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - - it: should schedule onto the same nodes as the gateway - template: migrations-job.yaml - set: - migrationJob: - enabled: true - nodeSelector: - karpenter.sh/nodepool: litellm-e2e - tolerations: - - key: workload - operator: Equal - value: litellm-e2e - effect: NoSchedule - asserts: - - equal: - path: spec.template.spec.nodeSelector - value: - karpenter.sh/nodepool: litellm-e2e - - equal: - path: spec.template.spec.tolerations - value: - - key: workload - operator: Equal - value: litellm-e2e - effect: NoSchedule - - - it: bounds the Job with a deadline by default, so a blocked migration cannot stall the release forever - set: - migrationJob: - enabled: true - asserts: - - equal: - path: spec.activeDeadlineSeconds - value: 1800 - - - it: honours an operator-supplied deadline - set: - migrationJob: - enabled: true - activeDeadlineSeconds: 600 - asserts: - - equal: - path: spec.activeDeadlineSeconds - value: 600 - - - it: omits the deadline entirely when it is nulled out, restoring the unbounded behaviour - set: - migrationJob: - enabled: true - activeDeadlineSeconds: null - asserts: - - notExists: - path: spec.activeDeadlineSeconds diff --git a/helm/litellm-helm/tests/pdb_tests.yaml b/helm/litellm-helm/tests/pdb_tests.yaml deleted file mode 100644 index 5e042e80bd3..00000000000 --- a/helm/litellm-helm/tests/pdb_tests.yaml +++ /dev/null @@ -1,45 +0,0 @@ -suite: "pdb enabled" -templates: - - poddisruptionbudget.yaml -tests: - - it: "renders a PDB with maxUnavailable=1" - set: - pdb.enabled: true - pdb.maxUnavailable: 1 - asserts: - - hasDocuments: { count: 1 } - - isKind: { of: PodDisruptionBudget } - - equal: { path: apiVersion, value: policy/v1 } - - equal: { path: spec.maxUnavailable, value: 1 } - - equal: - path: spec.selector.matchLabels - value: - app.kubernetes.io/name: litellm - app.kubernetes.io/instance: RELEASE-NAME - ---- -suite: "pdb disabled" -templates: - - poddisruptionbudget.yaml -tests: - - it: "does not render when disabled" - set: - pdb.enabled: false - asserts: - - hasDocuments: { count: 0 } - ---- -suite: "pdb minAvailable precedence" -templates: - - poddisruptionbudget.yaml -tests: - - it: "uses minAvailable when both are set" - set: - pdb.enabled: true - pdb.minAvailable: "50%" - pdb.maxUnavailable: 1 - asserts: - - isKind: { of: PodDisruptionBudget } - - equal: { path: apiVersion, value: policy/v1 } - - equal: { path: spec.minAvailable, value: "50%" } - - isNull: { path: spec.maxUnavailable } diff --git a/helm/litellm-helm/tests/service_tests.yaml b/helm/litellm-helm/tests/service_tests.yaml deleted file mode 100644 index 43ed0180bc8..00000000000 --- a/helm/litellm-helm/tests/service_tests.yaml +++ /dev/null @@ -1,116 +0,0 @@ -suite: Service Configuration Tests -templates: - - service.yaml -tests: - - it: should create a default ClusterIP service - template: service.yaml - asserts: - - isKind: - of: Service - - equal: - path: spec.type - value: ClusterIP - - equal: - path: spec.ports[0].port - value: 4000 - - equal: - path: spec.ports[0].targetPort - value: http - - equal: - path: spec.ports[0].protocol - value: TCP - - equal: - path: spec.ports[0].name - value: http - - isNull: - path: spec.loadBalancerClass - - - it: should create a NodePort service when specified - template: service.yaml - set: - service.type: NodePort - asserts: - - isKind: - of: Service - - equal: - path: spec.type - value: NodePort - - isNull: - path: spec.loadBalancerClass - - - it: should create a LoadBalancer service when specified - template: service.yaml - set: - service.type: LoadBalancer - asserts: - - isKind: - of: Service - - equal: - path: spec.type - value: LoadBalancer - - isNull: - path: spec.loadBalancerClass - - - it: should add loadBalancerClass when specified with LoadBalancer type - template: service.yaml - set: - service.type: LoadBalancer - service.loadBalancerClass: tailscale - asserts: - - isKind: - of: Service - - equal: - path: spec.type - value: LoadBalancer - - equal: - path: spec.loadBalancerClass - value: tailscale - - - it: should not add loadBalancerClass when specified with ClusterIP type - template: service.yaml - set: - service.type: ClusterIP - service.loadBalancerClass: tailscale - asserts: - - isKind: - of: Service - - equal: - path: spec.type - value: ClusterIP - - isNull: - path: spec.loadBalancerClass - - - it: should use custom port when specified - template: service.yaml - set: - service.port: 8080 - asserts: - - equal: - path: spec.ports[0].port - value: 8080 - - - it: should add service annotations when specified - template: service.yaml - set: - service.annotations: - cloud.google.com/load-balancer-type: "Internal" - service.beta.kubernetes.io/aws-load-balancer-internal: "true" - asserts: - - isKind: - of: Service - - equal: - path: metadata.annotations - value: - cloud.google.com/load-balancer-type: "Internal" - service.beta.kubernetes.io/aws-load-balancer-internal: "true" - - - it: should use the correct selector labels - template: service.yaml - asserts: - - isNotNull: - path: spec.selector - - equal: - path: spec.selector - value: - app.kubernetes.io/name: litellm - app.kubernetes.io/instance: RELEASE-NAME diff --git a/helm/litellm-helm/values.yaml b/helm/litellm-helm/values.yaml deleted file mode 100644 index fcee331a5aa..00000000000 --- a/helm/litellm-helm/values.yaml +++ /dev/null @@ -1,637 +0,0 @@ -# Default values for litellm. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 -# numWorkers: 2 - -image: - # Bundles the prisma CLI and engines, which is what lets the migrations job - # and the proxy's own schema check run without network access. - repository: ghcr.io/berriai/litellm - pullPolicy: Always - # Overrides the image tag whose default is the chart appVersion. - # tag: "latest" - tag: "" - -imagePullSecrets: [] -nameOverride: "litellm" -fullnameOverride: "" - -serviceAccount: - # Specifies whether a service account should be created - create: false - # Automatically mount a ServiceAccount's API credentials? - automount: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -# annotations for litellm deployment -deploymentAnnotations: {} -deploymentLabels: {} -deploymentMinReadySeconds: 0 - -# annotations for litellm pods -podAnnotations: {} -podLabels: {} - -# -- Deployment strategy configuration -# Example: -# type: RollingUpdate -# rollingUpdate: -# maxUnavailable: 0 -# maxSurge: 1 -strategy: {} - -terminationGracePeriodSeconds: 90 -topologySpreadConstraints: - [] - # - maxSkew: 1 - # topologyKey: kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - # labelSelector: - # matchLabels: - # app: litellm - -# At the time of writing, the litellm docker image requires write access to the -# filesystem on startup so that prisma can install some dependencies. -podSecurityContext: {} -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: false - # runAsNonRoot: true - # runAsUser: 1000 - -# A list of Kubernetes Secret objects that will be exported to the LiteLLM proxy -# pod as environment variables. These secrets can then be referenced in the -# configuration file (or "litellm" ConfigMap) with `os.environ/` -environmentSecrets: - [] - # - litellm-env-secret - -# A list of Kubernetes ConfigMap objects that will be exported to the LiteLLM proxy -# pod as environment variables. The ConfigMap kv-pairs can then be referenced in the -# configuration file (or "litellm" ConfigMap) with `os.environ/` -environmentConfigMaps: - [] - # - litellm-env-configmap - -service: - type: ClusterIP - port: 4000 - # If service type is `LoadBalancer` you can - # optionally specify loadBalancerClass - # loadBalancerClass: tailscale - -# Probes for LiteLLM gateway container -livenessProbe: - path: /health/liveliness - initialDelaySeconds: 0 - periodSeconds: 15 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 - -readinessProbe: - path: /health/readiness - initialDelaySeconds: 0 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 3 - -startupProbe: - path: /health/readiness - initialDelaySeconds: 0 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 30 - -ingress: - enabled: false - className: "nginx" - labels: {} - annotations: - {} - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: "true" - hosts: - - host: api.example.local - paths: - - path: / - pathType: ImplementationSpecific - tls: [] - # - secretName: chart-example-tls - # hosts: - # - chart-example.local - -# masterkey: changeit - -# if set, use this secret for the master key; otherwise, autogenerate a new one -masterkeySecretName: "" - -# if set, use this secret key for the master key; otherwise, use the default key -masterkeySecretKey: "" - -# Optional: enterprise billable-request metering. When enabled, the proxy counts -# successful requests to inference, MCP, and A2A endpoints and pushes them to -# LiteLLM's collector over mutual TLS. Requires an enterprise license. -# The client certificate identifies the deployment, so it is mounted read-only -# from an existing Secret and never passed through the environment. -billingMetrics: - enabled: false - endpoint: https://telemetry.litellm.ai # collector to push the counter to - secretName: litellm-billing-metrics-mtls # existing Secret holding tls.crt and tls.key - # Only for private or test collectors whose server certificate is not on the - # public web PKI. The production collector needs no CA override. - caSecretName: "" # existing Secret holding ca.crt - exportIntervalMs: "" # push cadence; the proxy defaults to 60000 - -proxyConfigMap: - # when true, creates a new configmap - create: true - # if create is false and name is set, use existing ConfigMap - # create: false - # name: "" - # key: "config.yaml" - -# The elements within proxy_config are rendered as config.yaml for the proxy -# Examples: https://github.com/BerriAI/litellm/tree/main/litellm/proxy/example_config_yaml -# Reference: https://docs.litellm.ai/docs/proxy/configs -proxy_config: - model_list: - # At least one model must exist for the proxy to start. - - model_name: gpt-3.5-turbo - litellm_params: - model: gpt-3.5-turbo - api_key: eXaMpLeOnLy - - model_name: fake-openai-endpoint - litellm_params: - model: openai/fake - api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ - general_settings: - master_key: os.environ/PROXY_MASTER_KEY - -# Serve Prometheus /metrics from a separate process (PROMETHEUS_METRICS_PORT) -# so a scrape never runs on an inference worker. Adds a `metrics` port to the -# container and a dedicated ClusterIP `-metrics` Service, and the -# ServiceMonitor scrapes it instead of the proxy port. The separate port has -# no virtual-key auth: keep it off public ingress. Needs the proxy image -# v1.101.0 or newer. -metricsServer: - enabled: false - port: 4001 - -# Opt-in sidecar that runs the post-response spend pipeline (cost calculation, -# spend logs, spend counters, budget reservation reconciliation) so the proxy's -# uvicorn workers only serialise a compact typed event and go back to serving -# inference. Same image and tag as the proxy, second container in the same pod, -# fed over loopback (a unix socket on a shared emptyDir, or 127.0.0.1 TCP). It -# reuses the pod's in-container pgbouncer (db.connectionPool) and the same Redis -# spend transaction buffer, so the per-pod DB connection budget is unchanged. -# Delivery is at-most-once inside the pod: events already handed to the sidecar -# are lost if it crashes before writing them; events the workers could not hand -# over follow onUnavailable. Both containers drain on SIGTERM within -# terminationGracePeriodSeconds -collector: - enabled: false - # unix:////.sock (the becomes a shared emptyDir) or tcp://127.0.0.1: - address: unix:///var/run/litellm/collector.sock - # Events each uvicorn worker holds in memory while the sidecar is slow or restarting - bufferSize: 1000 - # fallback: run the pipeline in the worker when the sidecar is unreachable or the - # buffer is full (spend stays exact, that request costs proxy CPU again) - # drop: count and discard the event instead (spend under-reports) - onUnavailable: fallback - # How long the workers keep pushing buffered events on shutdown, and how long the - # sidecar keeps serving its open connections after SIGTERM - drainTimeoutSeconds: 10 - command: - - python - - -m - - litellm.proxy.collector - # Sized independently of the proxy container; the pipeline is CPU bound - resources: {} - # requests: - # cpu: 500m - # memory: 1Gi - # limits: - # cpu: "1" - # memory: 2Gi - # When autoscaling.enabled, swap the pod-wide cpu Resource metric for an - # autoscaling/v2 ContainerResource metric on the proxy container only, so the - # sidecar's CPU never scales inference replicas. Needs Kubernetes 1.30+ (or the - # HPAContainerMetrics feature gate on 1.27 to 1.29) - scaleOnProxyContainerCpu: false - -resources: - {} - # Unset by default so the chart installs on small clusters such as Minikube, and so an - # upgrade never leaves a running pod Pending. Production deployments should set these. - # A proxy at DB-connected steady state needs about 1 CPU and 4Gi of memory per worker; - # sizing below that gets the pod OOMKilled once traffic and DB connections ramp up. - # Scale both figures with --num_workers, then uncomment the lines below and remove the - # curly braces after 'resources:'. See "Recommended Machine Specifications" in - # https://docs.litellm.ai/docs/proxy/prod. - # requests: - # cpu: "1" - # memory: 4Gi - # limits: - # cpu: "1" - # memory: 4Gi - -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - # 60 is the documented recommendation. See "Recommended Machine Specifications" - # in https://docs.litellm.ai/docs/proxy/prod. A new replica clears the startupProbe - # above only after up to failureThreshold x periodSeconds = 300 seconds, so a target - # high enough to trip near saturation adds capacity minutes after it was needed. - targetCPUUtilizationPercentage: 60 - # Deliberately left unset rather than given a value. The prisma query engine's - # resident memory is a high-water mark that ratchets to the pod's worst-ever write - # and is never returned, so a memory target reads the largest write a pod ever did - # rather than what it is doing now, and replicas ratchet up without scaling back in. - # Memory is a floor to provision under 'resources', not a signal to scale on. - # targetMemoryUtilizationPercentage: 80 - # behavior: {} - # Opt-in per-pod workload targets, rendered as autoscaling/v2 `Pods` metrics - # named `litellm_requests_per_second` and `litellm_tokens_per_second` with an - # AverageValue target, alongside whichever resource targets are set (the HPA - # follows the metric asking for the most replicas). A Prometheus Adapter must - # serve those two names on custom.metrics.k8s.io from the proxy's counters, - # grouped by the scrape target's `pod` label (enable serviceMonitor below so - # every pod is scraped on its own): - # litellm_requests_per_second: - # sum(rate(litellm_proxy_total_requests_metric_total{<<.LabelMatchers>>}[1m])) by (<<.GroupBy>>) - # litellm_tokens_per_second: - # sum(rate(litellm_total_tokens_metric_total{<<.LabelMatchers>>}[1m])) by (<<.GroupBy>>) - # rate() over [1m] is already per second, so no `* 60`. How fast the HPA - # reacts is set by that window, the scrape interval and the HPA sync period - # (15s by default), not by the unit: keep serviceMonitor.interval at 15s or - # faster so a 1m window holds at least 4 samples. averageValue takes SI - # suffixes, so "6M" is six million tokens per second per pod. Tokens are - # counted when a response completes, so TPS trails long streams. - targetRequestsPerSecond: "" - targetTokensPerSecond: "" - -# Autoscaling with keda is mutually exclusive with hpa -keda: - enabled: false - minReplicas: 1 - maxReplicas: 100 - pollingInterval: 30 - cooldownPeriod: 300 - # fallback: - # failureThreshold: 3 - # replicas: 11 - restoreToOriginalReplicaCount: false - scaledObject: - annotations: {} - triggers: [] - # - type: prometheus - # metadata: - # serverAddress: http://:9090 - # metricName: http_requests_total - # threshold: '100' - # query: sum(rate(http_requests_total{deployment="my-deployment"}[2m])) - # First-class Prometheus triggers on the proxy's own request and token - # counters, appended to `triggers`. Each target is the per-second load one - # replica should carry: KEDA divides the release-wide - # `sum(rate([1m]))` by it to pick the replica count. Thresholds - # are plain numbers (KEDA parses them as floats, no SI suffixes). The - # queries select samples by the release namespace and the `job` label the - # chart's ServiceMonitor produces (the metrics Service name), so enable - # serviceMonitor below together with metricsServer: the http port serves - # /metrics/ behind virtual-key auth and answers an unauthenticated scrape - # with 401. Reaction time comes from the [1m] window, the scrape interval - # and pollingInterval above, so keep both at 15s or faster. Tokens are - # counted at completion, so TPS trails long streams. serverAddress is - # required once either target is set. - prometheus: - serverAddress: "" - requestsPerSecond: "" - tokensPerSecond: "" - behavior: {} - # scaleDown: - # stabilizationWindowSeconds: 300 - # policies: - # - type: Pods - # value: 1 - # periodSeconds: 180 - # scaleUp: - # stabilizationWindowSeconds: 300 - # policies: - # - type: Pods - # value: 2 - # periodSeconds: 60 - -# Additional volumes on the output Deployment definition. -volumes: [] -# - name: foo -# secret: -# secretName: mysecret -# optional: false - -# Additional volumeMounts on the output Deployment definition. -volumeMounts: [] -# - name: foo -# mountPath: "/etc/foo" -# readOnly: true - -nodeSelector: {} - -tolerations: [] - -affinity: {} - -db: - # Use an existing postgres server/cluster - useExisting: false - - # How to connect to the existing postgres server/cluster - endpoint: localhost - database: litellm - url: postgresql://$(DATABASE_USERNAME):$(DATABASE_PASSWORD)@$(DATABASE_HOST)/$(DATABASE_NAME) - secret: - name: postgres - usernameKey: username - passwordKey: password - # Optional: when set, DATABASE_HOST will be sourced from this secret key instead of db.endpoint - endpointKey: "" - # Optional: when set, DATABASE_URL_READ_REPLICA will be sourced from this - # secret key instead of db.readReplicaUrl. Prefer this over the plain - # value: read-replica URLs typically embed credentials, and a value - # written to db.readReplicaUrl ends up visible in the rendered pod spec - # and the Helm release secret. - readReplicaUrlKey: "" - # Optional: when set, a DATABASE_READER_HOST env var is sourced from this - # secret key, so db.readReplicaUrl can compose the reader URL from - # individual secret components, e.g. - # postgresql://$(DATABASE_USERNAME):$(DATABASE_PASSWORD)@$(DATABASE_READER_HOST):5432/$(DATABASE_NAME) - # Use this when your secret store holds the bare reader hostname rather - # than a full connection URL. Only takes effect when readReplicaUrl is - # set; ignored when readReplicaUrlKey is set. - readReplicaEndpointKey: "" - - # Optional read-replica routing. When set, the proxy sends read-only - # queries (find_*, count, group_by, query_raw/_first) to this URL while - # writes continue to go to db.url. Useful for Aurora-style clusters with - # separate reader/writer endpoints. Leave empty to keep single-DB behavior. - # When IAM_TOKEN_DB_AUTH is enabled, the reader URL is auto-refreshed - # alongside the writer (host/port/user/db are parsed from this URL once - # at startup; only the IAM token rotates). - # - # If the URL embeds credentials, prefer db.secret.readReplicaUrlKey over - # this field — the plain value is rendered into the pod spec and the - # Helm release secret. This field is intended for credential-less URLs - # only (e.g. when IAM_TOKEN_DB_AUTH supplies the token at runtime). - readReplicaUrl: "" - - # In-container connection pool (PgBouncer, transaction mode) shared by every - # worker in the pod. Without it each --num_workers worker opens its own - # connection_limit connections to Postgres, so a pod's footprint against the - # database's connection ceiling is workers x connection_limit and grows with - # every replica. With it, the pod holds at most maxDbConnections upstream - # connections no matter how many workers run; the workers connect to the pool - # over loopback, with no extra network hop. Migrations still go straight to - # Postgres. Starting profile for numWorkers: 4 is maxDbConnections: 20, so - # a database with a 5000-connection ceiling fits roughly 200 replicas. - connectionPool: - enabled: false - maxDbConnections: 20 - maxClientConn: 1000 - - # Use the Stackgres Helm chart to deploy an instance of a Stackgres cluster. - # The Stackgres Operator must already be installed within the target - # Kubernetes cluster. - # TODO: Stackgres deployment currently unsupported - useStackgresOperator: false - - # Use the Postgres Helm chart to create a single node, stand alone postgres - # instance. See the "postgresql" top level key for additional configuration. - deployStandalone: true - -# Lifecycle hooks for the LiteLLM container -# -# Prefer the native /health/drain preStop hook over a fixed `sleep`: it marks -# the pod NotReady and blocks only until in-flight requests actually finish -# (bounded by GRACEFUL_SHUTDOWN_TIMEOUT, default 30s), instead of always -# waiting the worst-case duration. The drain runs once (the preStop hook and -# the SIGTERM handler share it), so set terminationGracePeriodSeconds a few -# seconds above GRACEFUL_SHUTDOWN_TIMEOUT to leave room for teardown before -# SIGKILL. -# -# /health/drain is off by default; enable it with -# general_settings.enable_drain_endpoint: true. The kubelet calls preStop -# hooks without proxy credentials, so when the health port is reachable from -# other pods (the common case) also set -# general_settings.drain_endpoint_token (or the DRAIN_ENDPOINT_TOKEN env -# var) and send the same value on the X-Drain-Token header from the hook. -# Calls missing/wrong the token get a 401 and have no side effect. -# Example: -# lifecycle: -# preStop: -# httpGet: -# path: /health/drain -# port: 4000 -# httpHeaders: -# - name: X-Drain-Token -# value: -lifecycle: {} - -# Settings for Bitnami postgresql chart (if db.deployStandalone is true, ignored -# otherwise) -# -# Bitnami retired the versioned tags under docker.io/bitnami and republished the -# archived builds under docker.io/bitnamilegacy, so the subchart's own image -# defaults no longer resolve. The repository below points at the same build the -# subchart was released with, which keeps the on-disk data directory layout -# identical for existing installs. -# -# Keep the tag pinned. docker.io/bitnami still publishes a floating `latest`, -# and starting a newer PostgreSQL major against an existing data directory -# leaves the server refusing to boot ("database files are incompatible with -# server") with no way back other than a dump taken beforehand. Crossing a major -# version is a dump-and-restore, not an image bump. The chart refuses to render -# an unpinned tag for this reason -postgresql: - architecture: standalone - image: - repository: bitnamilegacy/postgresql - tag: 16.2.0-debian-12-r6 - volumePermissions: - image: - repository: bitnamilegacy/os-shell - tag: 12-debian-12-r16 - metrics: - image: - repository: bitnamilegacy/postgres-exporter - tag: 0.15.0-debian-12-r14 - auth: - username: litellm - database: litellm - - # You should override these on the helm command line with - # `--set postgresql.auth.postgres-password=,postgresql.auth.password=` - password: NoTaGrEaTpAsSwOrD - postgres-password: NoTaGrEaTpAsSwOrD - - # A secret is created by this chart (litellm-helm) with the credentials that - # the new Postgres instance should use. - # existingSecret: "" - # secretKeys: - # userPasswordKey: password - -# Redis is the proxy's coordination store: cross-pod tpm/rpm rate limits, spend -# tracking, and the pod lock manager. Enabling this deploys the bundled Redis -# subchart, wires REDIS_HOST / REDIS_PORT / REDIS_PASSWORD into the proxy, and -# renders a `general_settings.coordination_redis` block into the proxy config. -# -# To point at an existing Redis instead, leave `enabled: false` and pass a -# secret for REDIS_HOST, REDIS_PORT, REDIS_PASSWORD or REDIS_URL; the proxy -# falls back to those env vars for coordination. Set `cache: true` in the proxy -# config only if you also want LLM response caching, which is independent of -# coordination -# -# When `redis.sentinel.enabled` is set, the coordination block is rendered with -# `sentinel_nodes` and `service_name` (from `redis.sentinel.masterSet`) instead -# of host/port, because a plain Redis client cannot talk to the sentinel port -# -# The image repositories carry the same bitnamilegacy repoint as postgresql -# above; the versioned tags the subchart ships with are gone from -# docker.io/bitnami -redis: - enabled: false - architecture: standalone - image: - repository: bitnamilegacy/redis - tag: 7.2.4-debian-12-r9 - sentinel: - image: - repository: bitnamilegacy/redis-sentinel - tag: 7.2.4-debian-12-r7 - metrics: - image: - repository: bitnamilegacy/redis-exporter - tag: 1.58.0-debian-12-r4 - volumePermissions: - image: - repository: bitnamilegacy/os-shell - tag: 12-debian-12-r16 - sysctl: - image: - repository: bitnamilegacy/os-shell - tag: 12-debian-12-r16 - kubectl: - image: - repository: bitnamilegacy/kubectl - tag: 1.29.2-debian-12-r3 - coordination: - # Set to false to keep the bundled Redis for response caching only and leave - # `general_settings.coordination_redis` out of the rendered config. A - # `coordination_redis` block you define yourself in `proxy_config` always wins - enabled: true - -# Prisma migration job settings -migrationJob: - enabled: true # Enable or disable the schema migration Job - retries: 3 # Number of retries for the Job in case of failure - backoffLimit: 4 # Backoff limit for Job restarts - # Wall-clock budget for the whole Job, shared across every `backoffLimit` - # retry rather than granted per attempt. Without it a migration that blocks - # on the database never fails, and when the Helm hook is enabled the release - # waits on it forever: `helm upgrade` and any GitOps controller driving it - # stop reconciling the whole chart until someone deletes the Job by hand. - # Set to null to opt out and restore the unbounded behaviour. - activeDeadlineSeconds: 1800 - disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0. - # Optional service account for the migration job. - # Only used when migrationJob.hooks.helm.enabled=true and serviceAccount.create=true. - # In that case, pre-install/pre-upgrade hooks run before normal resources, so this defaults to "default". - serviceAccountName: "" - annotations: {} - ttlSecondsAfterFinished: 120 - resources: {} - # Unset by default. This job runs the database migration and exits, so it does not - # need the steady-state headroom the proxy does; size it from your own migration - # runs rather than from the proxy figures above. - extraContainers: [] - extraInitContainers: [] - - # Hook configuration - hooks: - argocd: - enabled: true - helm: - enabled: false - -# Log level for the litellm proxy (sets LITELLM_LOG in the deployment env). -# Rendered as a direct `env:` entry, which in Kubernetes takes precedence over -# any `envFrom:` source. If you currently source LITELLM_LOG from an -# environmentSecret or environmentConfigMap, set `logLevel: ""` here to -# disable injection — otherwise this value silently overrides your secret / -# configmap entry. -# -# Setting LITELLM_LOG inside `envVars:` below also wins: the template skips -# this injection entirely when envVars already defines LITELLM_LOG. -logLevel: INFO - -# Additional environment variables to be added to the deployment as a map of key-value pairs -envVars: {} - -# USE_DDTRACE: "true" -# Additional environment variables to be added to the deployment as a list of k8s env vars -extraEnvVars: {} - -# if you want to override the container command, you can do so here -command: {} -# if you want to override the container args, you can do so here -args: {} - -# - name: EXTRA_ENV_VAR -# value: EXTRA_ENV_VAR_VALUE -# Additional Kubernetes resources to deploy with litellm -extraResources: [] - -# - apiVersion: v1 -# kind: ConfigMap -# metadata: -# name: my-extra-config -# data: -# foo: bar -# Pod Disruption Budget -pdb: - enabled: false - # Set exactly one of the following. If both are set, minAvailable takes precedence. - minAvailable: null # e.g. "50%" or 1 - maxUnavailable: null # e.g. 1 or "20%" - annotations: {} - labels: {} - -serviceMonitor: - enabled: false - labels: - {} - # test: test - annotations: - {} - # kubernetes.io/test: test - interval: 15s - scrapeTimeout: 10s - relabelings: [] - # - targetLabel: __meta_kubernetes_pod_node_name - # replacement: $1 - # action: replace - namespaceSelector: - matchNames: [] - # - test-namespace diff --git a/helm/litellm/.helmignore b/helm/litellm/.helmignore new file mode 100644 index 00000000000..ab8a25912fb --- /dev/null +++ b/helm/litellm/.helmignore @@ -0,0 +1,21 @@ +# Patterns to ignore when building packages. +.DS_Store +.git/ +.gitignore +.bzr/ +.bzrignore +.hg/ +.hgignore +.svn/ +*.swp +*.bak +*.tmp +*.orig +*~ +.project +.idea/ +*.tmproj +.vscode/ +/tests/ +/ci/ +README.md diff --git a/helm/litellm/Chart.yaml b/helm/litellm/Chart.yaml index e67f5790c7e..7f308a454f5 100644 --- a/helm/litellm/Chart.yaml +++ b/helm/litellm/Chart.yaml @@ -1,8 +1,8 @@ apiVersion: v2 name: litellm -description: LiteLLM componentized — gateway, UI backend, and UI as separate services +description: Deploys LiteLLM from one image, either componentized (gateway, UI backend, and UI as separate Deployments) or as a single monolith Deployment type: application -version: 0.1.0 -appVersion: "0.1.0" +version: 1.0.0 +appVersion: "1.104.0" annotations: org.opencontainers.image.source: "https://github.com/BerriAI/litellm" diff --git a/helm/litellm/README.md b/helm/litellm/README.md new file mode 100644 index 00000000000..2c3d4fdbdac --- /dev/null +++ b/helm/litellm/README.md @@ -0,0 +1,141 @@ +# litellm Helm chart + +Deploys [LiteLLM](https://github.com/BerriAI/litellm) from one image, `ghcr.io/berriai/litellm` (mirrored at `docker.litellm.ai/berriai/litellm`), in one of two layouts: + +- componentized (default): the `gateway` (LLM data plane, port 4000), the `backend` (management API, port 4001) and the `ui` (static dashboard, port 3000) each get their own Deployment, Service, HPA and PDB so they scale independently +- monolith: `monolith.enabled: true` renders one Deployment and Service running the full proxy (`args: [proxy, ...]`), which serves the gateway routes, the management routes and the Admin UI from a single process + +Both layouts run the same image. The image entrypoint dispatches on the first container argument (`proxy`, `gateway`, `backend`, `ui`, `migrations`, `metrics`, `collector`), so the chart only ever sets `args` and never `command` + +## Requirements + +Kubernetes 1.25+ and Helm 3.8+. The chart has no subcharts: bring your own PostgreSQL (`database.writer.*`) and, optionally, Redis (`redis.*`). The proxy also needs a master key, either an existing Secret named by `masterKey.secretName` or one the chart generates with `masterKey.generate: true` and `masterKey.secretName: ""` + +## Install + +```bash +kubectl create secret generic litellm-master-key-secret --from-literal=master-key=sk-change-me +helm install litellm helm/litellm \ + --set database.writer.host=postgres.example.com \ + --set database.writer.dbname=litellm \ + --set database.writer.passwordSecret.name=litellm-db-secret +``` + +The image tag defaults to the chart's `appVersion`. Override it with `image.tag`, or pin the bytes with `image.digest`, which renders as `repository:tag@digest` + +## Monolith quickstart + +```bash +helm install litellm helm/litellm \ + --set monolith.enabled=true \ + --set masterKey.secretName="" \ + --set masterKey.generate=true \ + --set database.writer.host=postgres.example.com \ + --set database.writer.dbname=litellm \ + --set database.writer.passwordSecret.name=litellm-db-secret +kubectl port-forward svc/litellm-litellm 4000:4000 +``` + +With `monolith.enabled: true`: + +- one Deployment and one Service named `-litellm` render, running `args: [proxy, --port, 4000, --config, /app/config/config.yaml]` plus `monolith.extraArgs` +- the gateway, backend and ui Deployments, Services, HPAs, PDBs and ServiceMonitors are not rendered, whatever `gateway.enabled`, `backend.enabled` and `ui.enabled` say +- every `gateway.*` value configures the monolith pod: `config`, `numWorkers`, `resources`, probes, `securityContext`, `hpa`, `keda`, `pdb`, `metricsServer`, `collector`, `volumes`, `extraEnv`, scheduling. The monolith runs as `serviceAccounts.gateway` +- every `backend.*` and `ui.*` value is ignored +- the Ingress sends every path, built in or from `ingress.extraPaths`, to the monolith Service +- the migrations Job renders exactly as in componentized mode + +## Testing + +```bash +helm lint helm/litellm +helm unittest -f 'tests/*.yaml' helm/litellm +helm test --logs +``` + +## Migrating from the litellm-helm chart + +The `litellm-helm` chart (`oci://ghcr.io/berriai/litellm-helm`) is retired; its published packages stay available for a grace period. Its flat values described one monolith Deployment, so the equivalent install here is `monolith.enabled: true` with the values moved under `gateway.*`. The per-component `gateway.image`, `backend.image`, `ui.image` and `migrations.image` blocks of earlier `helm/litellm` versions are gone too: the chart is a major bump to `1.0.0` and every container uses the top-level `image` + +| litellm-helm value | litellm value | +|---|---| +| (implicit single Deployment) | `monolith.enabled: true` | +| `image.repository` / `image.tag` / `image.pullPolicy` | `image.repository` / `image.tag` / `image.pullPolicy` (`image.digest` is new) | +| `replicaCount` | `gateway.replicaCount` | +| `args` | `monolith.extraArgs` (appended after the chart's proxy arguments) | +| `command` | removed: the image entrypoint dispatcher must stay in place | +| `proxy_config` | `gateway.config.proxy_config` | +| `proxyConfigMap.create: false` + `proxyConfigMap.name` | `gateway.config.create: false` and mount your ConfigMap with `gateway.volumes` / `gateway.volumeMounts`, or pass `--config` in `monolith.extraArgs` | +| `masterkeySecretName` / `masterkeySecretKey` | `masterKey.secretName` / `masterKey.secretKey` | +| `masterkeySecretName: ""` (auto generated Secret) | `masterKey.generate: true` with `masterKey.secretName: ""` (creates a new key; copy the old key first, see below, to keep existing credentials valid) | +| `db.useExisting`, `db.endpoint`, `db.database`, `db.secret.*` | `database.writer.host`, `database.writer.port`, `database.writer.dbname`, `database.writer.passwordSecret.*` | +| `db.readReplicaUrl` / `db.secret.readReplica*` | `database.reader.*` | +| `db.connectionPool.*` | `database.connectionPool.*` | +| `db.deployStandalone: true` / `postgresql.*` | removed: the chart ships no PostgreSQL, point `database.writer.*` at your own | +| `redis.enabled: true` (bundled) | removed: the chart ships no Redis, point `redis.host` at your own | +| external Redis via `envVars` | `redis.host`, `redis.port`, `redis.passwordSecret.*`, `redis.cluster` | +| `envVars` / `extraEnvVars` | `gateway.extraEnv` (list of `name` / `value` or `valueFrom` entries) | +| `environmentSecrets` | `gateway.envSecrets` | +| `environmentConfigMaps` | `gateway.envConfigMaps` | +| `logLevel` | `gateway.logLevel` | +| `resources` | `gateway.resources` | +| `livenessProbe` / `readinessProbe` / `startupProbe` | `gateway.livenessProbe` / `gateway.readinessProbe` / `gateway.startupProbe` | +| `securityContext` / `podSecurityContext` | `gateway.securityContext` / `gateway.podSecurityContext` | +| `service.*` | `gateway.service.*` | +| `ingress.*` | `ingress.*` (routes to the monolith Service in monolith mode) | +| `autoscaling.*` | `gateway.hpa.*` | +| `keda.*` | `gateway.keda.*` (`keda.prometheus.requestsPerSecond` / `tokensPerSecond` are `gateway.keda.prometheus.targetRequestsPerSecond` / `targetTokensPerSecond`) | +| `pdb.*` | `gateway.pdb.*` | +| `metricsServer.*` | `gateway.metricsServer.*` | +| `serviceMonitor.*` | `gateway.serviceMonitor.*` | +| `collector.*` | `gateway.collector.*` | +| `billingMetrics.*` | `billingMetrics.*` | +| `migrationJob.*` | `migrationJob.*` | +| `volumes` / `volumeMounts` | `gateway.volumes` / `gateway.volumeMounts` | +| `extraContainers` / `extraInitContainers` | `gateway.extraContainers` / `gateway.extraInitContainers` | +| `lifecycle` | `gateway.lifecycle` | +| `strategy` | `gateway.strategy` | +| `deploymentAnnotations` / `deploymentLabels` / `deploymentMinReadySeconds` | `gateway.deploymentAnnotations` / `gateway.deploymentLabels` / `gateway.minReadySeconds` | +| `podAnnotations` / `podLabels` | `gateway.podAnnotations` / `gateway.podLabels` | +| `nodeSelector` / `tolerations` / `affinity` / `topologySpreadConstraints` | `gateway.nodeSelector` / `gateway.tolerations` / `gateway.affinity` / `gateway.topologySpreadConstraints` | +| `terminationGracePeriodSeconds` | `gateway.terminationGracePeriodSeconds` | +| `serviceAccount.*` | `serviceAccounts.gateway.*` | +| `extraResources` | `extraResources` | +| `nameOverride: "litellm"` | `nameOverride: ""` (the chart name is already `litellm`) | + +A minimal migration: + +```yaml +monolith: + enabled: true +masterKey: + secretName: litellm-master-key-secret +database: + writer: + host: postgres.example.com + dbname: litellm + passwordSecret: + name: litellm-db-secret + usernameKey: username + passwordKey: password +gateway: + replicaCount: 2 + config: + proxy_config: + model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + envSecrets: + - litellm-provider-keys +``` + +The monolith Service keeps the `-litellm` name the old chart produced through its `nameOverride: "litellm"` default, so an existing Ingress or port-forward keeps working after `helm uninstall` of the old release and `helm install` of this one. The generated Secret in this chart has `helm.sh/resource-policy: keep` and is reused across upgrades of this chart. The retired chart stored its generated key under a different Secret and data key, and `helm uninstall` deletes that Secret, so copy it into a Secret you own before uninstalling the old release to keep existing credentials valid, for example: + +```bash +kubectl create secret generic litellm-master-key-secret \ + --from-literal=master-key="$(kubectl get secret -litellm-masterkey -o jsonpath='{.data.masterkey}' | base64 -d)" +``` + +Then set `masterKey.secretName: litellm-master-key-secret`, as in the minimal example above. Releases that set `masterkeySecretName` already point `masterKey.secretName` and `masterKey.secretKey` at that Secret diff --git a/helm/litellm/ci/postgres.yaml b/helm/litellm/ci/postgres.yaml new file mode 100644 index 00000000000..aa210e932b0 --- /dev/null +++ b/helm/litellm/ci/postgres.yaml @@ -0,0 +1,58 @@ +# Throwaway PostgreSQL for the CircleCI helm_chart_testing job. Applied with +# kubectl before `helm install`; the chart itself ships no database. +apiVersion: v1 +kind: Secret +metadata: + name: litellm-ci-postgres +type: Opaque +stringData: + username: litellm + password: litellm-ci +--- +apiVersion: v1 +kind: Service +metadata: + name: litellm-ci-postgres +spec: + selector: + app: litellm-ci-postgres + ports: + - port: 5432 + targetPort: 5432 +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: litellm-ci-postgres +spec: + replicas: 1 + selector: + matchLabels: + app: litellm-ci-postgres + template: + metadata: + labels: + app: litellm-ci-postgres + spec: + containers: + - name: postgres + image: postgres:16.10-alpine@sha256:029660641a0cfc575b14f336ba448fb8a75fd595d42e1fa316b9fb4378742297 + ports: + - containerPort: 5432 + env: + - name: POSTGRES_DB + value: litellm + - name: POSTGRES_USER + valueFrom: + secretKeyRef: + name: litellm-ci-postgres + key: username + - name: POSTGRES_PASSWORD + valueFrom: + secretKeyRef: + name: litellm-ci-postgres + key: password + readinessProbe: + exec: + command: [pg_isready, -U, litellm, -d, litellm] + periodSeconds: 2 diff --git a/helm/litellm/ci/test-values.yaml b/helm/litellm/ci/test-values.yaml new file mode 100644 index 00000000000..40c68489ed3 --- /dev/null +++ b/helm/litellm/ci/test-values.yaml @@ -0,0 +1,34 @@ +# Values for the CircleCI helm_chart_testing job: a monolith install on a +# kind cluster from the image built by the pipeline (image.repository / tag / +# pullPolicy are passed with --set), backed by the throwaway PostgreSQL in +# ci/postgres.yaml so the migrations Job and the proxy have a database. +monolith: + enabled: true + +masterKey: + secretName: "" + generate: true + +gateway: + replicaCount: 1 + hpa: + enabled: false + resources: + requests: + cpu: 250m + memory: 512Mi + limits: + memory: 2Gi + config: + create: true + proxy_config: + general_settings: {} + +database: + writer: + host: litellm-ci-postgres + dbname: litellm + passwordSecret: + name: litellm-ci-postgres + usernameKey: username + passwordKey: password diff --git a/helm/litellm/templates/NOTES.txt b/helm/litellm/templates/NOTES.txt index 468cf621b32..554ab072f2b 100644 --- a/helm/litellm/templates/NOTES.txt +++ b/helm/litellm/templates/NOTES.txt @@ -1,4 +1,16 @@ -LiteLLM componentized — release {{ .Release.Name }} in namespace {{ .Release.Namespace }}. +{{- if .Values.monolith.enabled }} +LiteLLM monolith: release {{ .Release.Name }} in namespace {{ .Release.Namespace }}, image {{ include "litellm.image" . }} + +One Deployment runs the full proxy (gateway routes, management routes and the Admin UI): + - proxy : Service {{ include "litellm.workload.fullname" . }} on port {{ .Values.gateway.service.port }} + +Port-forward: + kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.workload.fullname" . }} {{ .Values.gateway.service.port }} + +The monolith takes its configuration from the gateway.* values (config, resources, probes, hpa, pdb, keda, +metricsServer, collector, volumes, env, scheduling). backend.* and ui.* are ignored while monolith.enabled is true +{{- else }} +LiteLLM componentized: release {{ .Release.Name }} in namespace {{ .Release.Namespace }}, image {{ include "litellm.image" . }} Components: {{- if .Values.gateway.enabled }} @@ -16,39 +28,44 @@ Port-forward examples: kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.backend.fullname" . }} {{ .Values.backend.service.port }} kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.ui.fullname" . }} {{ .Values.ui.service.port }} +Set monolith.enabled=true to run everything as one Deployment instead +{{- end }} + Reminders: - Sensitive values come from Secret references only. Before installing, set: - - masterKey.secretName (Secret with the proxy master key) + - masterKey.secretName (Secret with the proxy master key, or + masterKey.generate: true with secretName "" to let the chart create one) - database.writer.{host,port,dbname} (writer connection pieces) - database.writer.passwordSecret.{name,usernameKey,passwordKey} (Secret holding the writer DB username + password) - - database.writer.useIAMAuth: true (optional — chart sets IAM_TOKEN_DB_AUTH=true and + - database.writer.useIAMAuth: true (optional: chart sets IAM_TOKEN_DB_AUTH=true and omits DATABASE_PASSWORD / DATABASE_URL so the proxy mints the URL from an IAM token at startup) - - database.reader.host (optional — enables read-replica routing; reader + - database.reader.host (optional: enables read-replica routing; reader .passwordSecret.name is required when set, unless .useIAMAuth is true) - - database.reader.useIAMAuth: true (optional, requires database.writer.useIAMAuth: true — + - database.reader.useIAMAuth: true (optional, requires database.writer.useIAMAuth: true; chart emits DATABASE_*_READ_REPLICA env vars and omits DATABASE_PASSWORD_READ_REPLICA / DATABASE_URL_READ_REPLICA so the proxy mints the reader URL from an IAM token at startup) - - redis.passwordSecret.name (optional — set when redis.host is provided and the + - redis.host (optional: the proxy's coordination Redis) + - redis.passwordSecret.name (optional: set when redis.host is provided and the cache requires auth) - - redis.cluster: true (optional — chart sets REDIS_CLUSTER_NODES from + - redis.cluster: true (optional: chart sets REDIS_CLUSTER_NODES from redis.host / redis.port so the proxy's Cache() constructs a RedisClusterCache; the cluster client discovers remaining nodes from CLUSTER SLOTS) - - Per-component extras (gateway / backend / ui): - - {component}.extraEnv / envConfigMaps / envSecrets (the latter two are lists of resource names → - envFrom configMapRef / secretRef) + - Per-component extras (gateway / backend / ui; gateway.* also drives the monolith): + - {component}.extraEnv / envConfigMaps / envSecrets (the latter two are lists of resource names rendered + as envFrom configMapRef / secretRef) - {component}.logLevel (renders as LITELLM_LOG) - gateway.config.proxy_config (rendered into a ConfigMap and mounted at - /app/config/config.yaml; gateway reads it via - CONFIG_FILE_PATH) + /app/config/config.yaml; read via CONFIG_FILE_PATH) - {component}.pdb.{enabled,minAvailable,maxUnavailable} (per-component PodDisruptionBudget; disabled by - default — with hpa.minReplicas of 1, minAvailable: 1 + default: with hpa.minReplicas of 1, minAvailable: 1 would block node drains) - {component}.topologySpreadConstraints (standard k8s list, e.g. spread replicas across topology.kubernetes.io/zone) - - Enable ingress.enabled=true to dispatch / → ui, gateway data-plane prefixes → gateway, and the catch-all → backend. + - Enable ingress.enabled=true to dispatch / to ui, gateway data-plane prefixes to gateway and the catch-all to + backend; in monolith mode every path goes to the monolith Service diff --git a/helm/litellm/templates/_helpers.tpl b/helm/litellm/templates/_helpers.tpl index 20fd1a722dc..cb8a7bf6641 100644 --- a/helm/litellm/templates/_helpers.tpl +++ b/helm/litellm/templates/_helpers.tpl @@ -1,11 +1,132 @@ {{/* -Common naming + label helpers shared by gateway, backend, and ui templates. +Common naming + label helpers shared by gateway, backend, ui, and monolith templates. */}} {{- define "litellm.name" -}} {{- default .Chart.Name .Values.nameOverride | trunc 63 | trimSuffix "-" -}} {{- end -}} +{{/* +The one image reference every container in the chart uses. `repository:tag`, +or `repository:tag@digest` when image.digest is set; the tag falls back to +the chart appVersion. +*/}} +{{- define "litellm.image" -}} +{{- $tag := .Values.image.tag | default .Chart.AppVersion -}} +{{- if .Values.image.digest -}} +{{- printf "%s:%s@%s" .Values.image.repository $tag .Values.image.digest -}} +{{- else -}} +{{- printf "%s:%s" .Values.image.repository $tag -}} +{{- end -}} +{{- end -}} + +{{/* +Componentized mode renders the gateway / backend / ui Deployments only when +monolith mode is off. +*/}} +{{- define "litellm.gateway.render" -}} +{{- if and .Values.gateway.enabled (not .Values.monolith.enabled) -}}true{{- end -}} +{{- end -}} + +{{- define "litellm.backend.render" -}} +{{- if and .Values.backend.enabled (not .Values.monolith.enabled) -}}true{{- end -}} +{{- end -}} + +{{- define "litellm.ui.render" -}} +{{- if and .Values.ui.enabled (not .Values.monolith.enabled) -}}true{{- end -}} +{{- end -}} + +{{/* +The proxy Deployment (monolith mode) and the gateway Deployment +(componentized mode) are the same pod spec fed by .Values.gateway, so the +templates that serve both (HPA, KEDA, PDB, metrics Service, ServiceMonitor) +resolve their name, component label and selector through these. +*/}} +{{- define "litellm.proxy.fullname" -}} +{{- include "litellm.fullname" . -}} +{{- end -}} + +{{- define "litellm.proxy.selectorLabels" -}} +app.kubernetes.io/name: {{ include "litellm.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +app.kubernetes.io/component: proxy +{{- end -}} + +{{- define "litellm.workload.componentName" -}} +{{- if .Values.monolith.enabled -}}proxy{{- else -}}gateway{{- end -}} +{{- end -}} + +{{- define "litellm.workload.fullname" -}} +{{- if .Values.monolith.enabled -}} +{{- include "litellm.proxy.fullname" . -}} +{{- else -}} +{{- include "litellm.gateway.fullname" . -}} +{{- end -}} +{{- end -}} + +{{- define "litellm.workload.selectorLabels" -}} +{{- if .Values.monolith.enabled -}} +{{- include "litellm.proxy.selectorLabels" . -}} +{{- else -}} +{{- include "litellm.gateway.selectorLabels" . -}} +{{- end -}} +{{- end -}} + +{{- define "litellm.workload.render" -}} +{{- if or .Values.monolith.enabled .Values.gateway.enabled -}}true{{- end -}} +{{- end -}} + +{{/* +Container arguments for the proxy (monolith) container. The image entrypoint +dispatches on the first argument. +*/}} +{{- define "litellm.proxy.args" -}} +- proxy +- --port +- "4000" +{{- if .Values.gateway.config.create }} +- --config +- /app/config/config.yaml +{{- end }} +{{- with .Values.monolith.extraArgs }} +{{ toYaml . }} +{{- end }} +{{- end -}} + +{{/* +Master key Secret reference. `masterKey.secretName` when set, otherwise the +Secret the chart generates when `masterKey.generate` is true. +*/}} +{{- define "litellm.masterKey.generatedSecretName" -}} +{{- printf "%s-masterkey" (include "litellm.fullname" .) -}} +{{- end -}} + +{{- define "litellm.masterKey.secretName" -}} +{{- if .Values.masterKey.secretName -}} +{{- .Values.masterKey.secretName -}} +{{- else if .Values.masterKey.generate -}} +{{- include "litellm.masterKey.generatedSecretName" . -}} +{{- else -}} +{{- fail "masterKey.secretName is required (the chart never accepts an inline master key); set it to an existing Secret or set masterKey.generate: true" -}} +{{- end -}} +{{- end -}} + +{{/* +Writer connection pieces as a dict (host, port, dbname, passwordSecret, ...) +so `litellm.serverEnv` renders them in one place. +*/}} +{{- define "litellm.database.writer" -}} +{{- toYaml .Values.database.writer -}} +{{- end -}} + +{{/* +Coordination Redis pieces: host, port, cluster and the passwordSecret pair. +An empty host means no Redis. +*/}} +{{- define "litellm.redis.connection" -}} +{{- toYaml (dict "host" .Values.redis.host "port" .Values.redis.port "cluster" .Values.redis.cluster "passwordSecret" .Values.redis.passwordSecret) -}} +{{- end -}} + {{- define "litellm.fullname" -}} {{- if .Values.fullnameOverride -}} {{- .Values.fullnameOverride | trunc 63 | trimSuffix "-" -}} @@ -113,6 +234,7 @@ Each component (gateway, backend, ui) has its own SA config under .Values.serviceAccounts.. When `create` is true and `name` is empty the chart defaults to "-litellm-". When `create` is false the chart uses the provided name, or the namespace `default` SA. +The monolith pod runs as the gateway ServiceAccount. */}} {{- define "litellm.gateway.serviceAccountName" -}} {{- if .Values.serviceAccounts.gateway.create -}} @@ -191,6 +313,19 @@ by the controller rather than declared, so nothing there can collide. {{- toYaml .podLabels }} {{- end -}} +{{/* +LITELLM_MASTER_KEY for the app containers only. The migrations Job runs as a +pre-install hook, before the chart's generated master key Secret exists, and +migrations/run.py never reads the key, so the Job must not reference it. +*/}} +{{- define "litellm.masterKeyEnv" -}} +- name: LITELLM_MASTER_KEY + valueFrom: + secretKeyRef: + name: {{ include "litellm.masterKey.secretName" . }} + key: {{ .Values.masterKey.secretKey | default "master-key" }} +{{- end -}} + {{/* Master-key + database + redis env block — shared by gateway, backend, and the migrations Job. @@ -232,16 +367,11 @@ IAM_TOKEN_DB_AUTH / AZURE_POSTGRESQL_AUTH toggle that only the writer sets. {{- define "litellm.serverEnv" -}} {{- $root := .root -}} {{- $component := .component -}} -- name: LITELLM_MASTER_KEY - valueFrom: - secretKeyRef: - name: {{ required "masterKey.secretName is required (the chart no longer accepts an inline master key)" $root.Values.masterKey.secretName }} - key: {{ $root.Values.masterKey.secretKey | default "master-key" }} {{- if $component.logLevel }} - name: LITELLM_LOG value: {{ $component.logLevel | quote }} {{- end }} -{{- with $root.Values.database.writer }} +{{- with (fromYaml (include "litellm.database.writer" $root)) }} - name: DATABASE_HOST value: {{ required "database.writer.host is required" .host | quote }} - name: DATABASE_PORT @@ -251,6 +381,11 @@ IAM_TOKEN_DB_AUTH / AZURE_POSTGRESQL_AUTH toggle that only the writer sets. secretKeyRef: name: {{ required "database.writer.passwordSecret.name is required" .passwordSecret.name }} key: {{ .passwordSecret.usernameKey | default "username" }} +- name: DATABASE_USERNAME + valueFrom: + secretKeyRef: + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.usernameKey | default "username" }} - name: DATABASE_NAME value: {{ required "database.writer.dbname is required" .dbname | quote }} {{- if .schema }} @@ -341,26 +476,28 @@ harmless no-op for the Job and authoritative for the app pods. tracking, pod lock manager) via its REDIS_* env fallback. An explicit `general_settings.coordination_redis` block in proxy_config takes precedence over anything emitted here. */}} -{{- if $root.Values.redis.host }} +{{- with (fromYaml (include "litellm.redis.connection" $root)) }} +{{- if .host }} - name: REDIS_HOST - value: {{ $root.Values.redis.host | quote }} + value: {{ .host | quote }} - name: REDIS_PORT - value: {{ $root.Values.redis.port | quote }} -{{- if $root.Values.redis.passwordSecret.name }} + value: {{ .port | quote }} +{{- if .passwordSecret.name }} - name: REDIS_PASSWORD valueFrom: secretKeyRef: - name: {{ $root.Values.redis.passwordSecret.name }} - key: {{ $root.Values.redis.passwordSecret.passwordKey | default "password" }} + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.passwordKey | default "password" }} {{- end }} -{{- if $root.Values.redis.cluster }} +{{- if .cluster }} {{/* The proxy falls back to REDIS_CLUSTER_NODES (JSON) to build a cluster-mode coordination client when `general_settings.coordination_redis` is absent and no plain-Redis response cache is configured. We seed with the single configured endpoint; the cluster client discovers the remaining nodes from CLUSTER SLOTS at startup. */}} - name: REDIS_CLUSTER_NODES - value: {{ printf "[{\"host\":%q,\"port\":%v}]" $root.Values.redis.host (int $root.Values.redis.port) | quote }} + value: {{ printf "[{\"host\":%q,\"port\":%v}]" .host (int .port) | quote }} +{{- end }} {{- end }} {{- end }} {{- with $component.extraEnv }} @@ -369,7 +506,7 @@ harmless no-op for the Job and authoritative for the app pods. {{- end -}} {{/* -In-container PgBouncer env for the gateway container. Under IAM or Entra auth the pooler mints and renews the database token itself. +In-container PgBouncer env for the gateway and proxy containers. Under IAM or Entra auth the pooler mints and renews the database token itself. */}} {{- define "litellm.connectionPoolEnv" -}} {{- with .Values.database.connectionPool -}} @@ -406,7 +543,7 @@ than silently replaced by the fallback. {{- $max := $component.pdb.maxUnavailable -}} {{- $minSet := not (or (kindIs "invalid" $min) (eq (printf "%v" $min) "")) -}} {{- $maxSet := not (or (kindIs "invalid" $max) (eq (printf "%v" $max) "")) -}} -{{- if and $component.enabled $component.pdb $component.pdb.enabled }} +{{- if and .enabled $component.pdb $component.pdb.enabled }} apiVersion: policy/v1 kind: PodDisruptionBudget metadata: @@ -464,6 +601,16 @@ ImplementationSpecific {{- end -}} {{- end -}} +{{/* +Service spec fields shared by the component and monolith Services. +Invoke with the component's `service` dict. +*/}} +{{- define "litellm.service.extras" -}} +{{- if and (eq .type "LoadBalancer") .loadBalancerClass -}} +loadBalancerClass: {{ .loadBalancerClass | quote }} +{{- end }} +{{- end -}} + {{- define "litellm.gateway.prometheusMultiprocDir" -}}/tmp/litellm_prometheus_multiproc{{- end -}} {{/* diff --git a/helm/litellm/templates/_workload.tpl b/helm/litellm/templates/_workload.tpl new file mode 100644 index 00000000000..87bf9f6812d --- /dev/null +++ b/helm/litellm/templates/_workload.tpl @@ -0,0 +1,487 @@ +{{/* +The pod that runs the LLM data plane. In componentized mode it is the gateway +Deployment (`args: [gateway, ...]`), in monolith mode the proxy Deployment +(`args: [proxy, ...]`) that also serves the management API and the Admin UI. +Both are configured by .Values.gateway; the name, component label, selector +and args come from the litellm.workload.* helpers. +*/}} +{{- define "litellm.workload.deployment" -}} +{{- $component := include "litellm.workload.componentName" . -}} +{{- $fullname := include "litellm.workload.fullname" . -}} +{{- if and .Values.gateway.hpa.enabled .Values.gateway.keda.enabled }} +{{- fail "gateway.hpa.enabled and gateway.keda.enabled are mutually exclusive: two autoscalers on one Deployment fight over the replica count, so set gateway.hpa.enabled: false when using KEDA" }} +{{- end }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ $fullname }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} + {{- with .Values.gateway.deploymentLabels }} + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with .Values.gateway.deploymentAnnotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + {{- if and (not .Values.gateway.hpa.enabled) (not .Values.gateway.keda.enabled) (not (kindIs "invalid" .Values.gateway.replicaCount)) }} + replicas: {{ .Values.gateway.replicaCount }} + {{- end }} + {{- $minReady := .Values.gateway.minReadySeconds }} + {{- if not (or (kindIs "invalid" $minReady) (eq (printf "%v" $minReady) "")) }} + minReadySeconds: {{ $minReady }} + {{- end }} + {{- with .Values.gateway.strategy }} + strategy: + {{- toYaml . | nindent 4 }} + {{- end }} + selector: + matchLabels: + {{- include "litellm.workload.selectorLabels" . | nindent 6 }} + template: + metadata: + annotations: + {{- if .Values.gateway.config.create }} + checksum/config: {{ include (print $.Template.BasePath "/gateway/configmap.yaml") . | sha256sum }} + {{- end }} + {{- with .Values.gateway.podAnnotations }} + {{- toYaml . | nindent 8 }} + {{- end }} + labels: + {{- include "litellm.workload.selectorLabels" . | nindent 8 }} + {{- with .Values.gateway.podLabels }} + {{- include "litellm.podLabels" (dict "podLabels" . "componentName" "gateway") | nindent 8 }} + {{- end }} + spec: + serviceAccountName: {{ include "litellm.gateway.serviceAccountName" . }} + automountServiceAccountToken: {{ .Values.serviceAccounts.gateway.automount }} + {{- with .Values.gateway.podSecurityContext }} + securityContext: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.extraInitContainers }} + initContainers: + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} + containers: + - name: {{ $component }} + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} + {{- with .Values.gateway.securityContext }} + securityContext: + {{- toYaml . | nindent 12 }} + {{- end }} + args: + {{- if .Values.monolith.enabled }} + {{- include "litellm.proxy.args" . | nindent 12 }} + {{- else }} + - gateway + - --host + - 0.0.0.0 + - --port + - "4000" + {{- end }} + ports: + - name: http + containerPort: 4000 + protocol: TCP + env: + {{- include "litellm.masterKeyEnv" $ | nindent 12 }} + {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.gateway) | nindent 12 }} + {{- if .Values.gateway.config.create }} + - name: CONFIG_FILE_PATH + value: /app/config/config.yaml + {{- end }} + {{- if .Values.gateway.numWorkers }} + - name: NUM_WORKERS + value: {{ .Values.gateway.numWorkers | quote }} + {{- end }} + {{- if .Values.database.connectionPool.enabled }} + {{- include "litellm.connectionPoolEnv" $ | nindent 12 }} + {{- end }} + {{- if .Values.billingMetrics.enabled }} + {{- include "litellm.billingMetricsEnv" . | nindent 12 }} + {{- end }} + {{- if .Values.gateway.metricsServer.enabled }} + {{- if eq (int .Values.gateway.metricsServer.port) 4000 }} + {{- fail (printf "gateway.metricsServer.port must differ from the %s port 4000" $component) }} + {{- end }} + - name: PROMETHEUS_MULTIPROC_DIR + value: {{ include "litellm.gateway.prometheusMultiprocDir" . }} + {{- end }} + {{- if .Values.gateway.collector.enabled }} + {{- include "litellm.gateway.collectorEnv" . | nindent 12 }} + {{- end }} + {{- include "litellm.envFrom" .Values.gateway | nindent 10 }} + {{- if or .Values.gateway.config.create .Values.gateway.volumeMounts .Values.billingMetrics.enabled .Values.gateway.metricsServer.enabled (include "litellm.gateway.collectorSocketDir" .) }} + volumeMounts: + {{- if .Values.gateway.config.create }} + - name: gateway-config + mountPath: /app/config/config.yaml + subPath: config.yaml + {{- end }} + {{- if .Values.gateway.metricsServer.enabled }} + - name: prometheus-multiproc + mountPath: {{ include "litellm.gateway.prometheusMultiprocDir" . }} + {{- end }} + {{- if include "litellm.gateway.collectorSocketDir" . }} + - name: collector-socket + mountPath: {{ include "litellm.gateway.collectorSocketDir" . }} + {{- end }} + {{- if .Values.billingMetrics.enabled }} + {{- include "litellm.billingMetricsVolumeMounts" . | nindent 12 }} + {{- end }} + {{- with .Values.gateway.volumeMounts }} + {{- toYaml . | nindent 12 }} + {{- end }} + {{- end }} + {{- with .Values.gateway.livenessProbe }} + livenessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.gateway.readinessProbe }} + readinessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.gateway.startupProbe }} + startupProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.gateway.lifecycle }} + lifecycle: + {{- toYaml . | nindent 12 }} + {{- end }} + resources: + {{- toYaml .Values.gateway.resources | nindent 12 }} + {{- if .Values.gateway.metricsServer.enabled }} + - name: metrics + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} + {{- with .Values.gateway.securityContext }} + securityContext: + {{- toYaml . | nindent 12 }} + {{- end }} + args: + - metrics + - --port + - {{ .Values.gateway.metricsServer.port | quote }} + env: + - name: PROMETHEUS_MULTIPROC_DIR + value: {{ include "litellm.gateway.prometheusMultiprocDir" . }} + ports: + - name: metrics + containerPort: {{ .Values.gateway.metricsServer.port }} + protocol: TCP + volumeMounts: + - name: prometheus-multiproc + mountPath: {{ include "litellm.gateway.prometheusMultiprocDir" . }} + readinessProbe: + tcpSocket: { port: metrics } + periodSeconds: 10 + livenessProbe: + tcpSocket: { port: metrics } + periodSeconds: 15 + failureThreshold: 6 + resources: + {{- toYaml .Values.gateway.metricsServer.resources | nindent 12 }} + {{- end }} + {{- if .Values.gateway.collector.enabled }} + - name: collector + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} + {{- with .Values.gateway.securityContext }} + securityContext: + {{- toYaml . | nindent 12 }} + {{- end }} + args: + - collector + env: + {{- include "litellm.masterKeyEnv" $ | nindent 12 }} + {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.gateway) | nindent 12 }} + {{- if .Values.gateway.config.create }} + - name: CONFIG_FILE_PATH + value: /app/config/config.yaml + {{- end }} + {{- if .Values.database.connectionPool.enabled }} + {{- include "litellm.connectionPoolEnv" $ | nindent 12 }} + {{- end }} + {{- include "litellm.gateway.collectorEnv" . | nindent 12 }} + - name: LITELLM_JOB_ROLE + value: collector + {{- include "litellm.envFrom" .Values.gateway | nindent 10 }} + {{- if or .Values.gateway.config.create .Values.gateway.volumeMounts (include "litellm.gateway.collectorSocketDir" .) }} + volumeMounts: + {{- if .Values.gateway.config.create }} + - name: gateway-config + mountPath: /app/config/config.yaml + subPath: config.yaml + {{- end }} + {{- if include "litellm.gateway.collectorSocketDir" . }} + - name: collector-socket + mountPath: {{ include "litellm.gateway.collectorSocketDir" . }} + {{- end }} + {{- with .Values.gateway.volumeMounts }} + {{- toYaml . | nindent 12 }} + {{- end }} + {{- end }} + resources: + {{- toYaml .Values.gateway.collector.resources | nindent 12 }} + {{- end }} + {{- with .Values.gateway.extraContainers }} + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} + {{- if or .Values.gateway.config.create .Values.gateway.volumes .Values.billingMetrics.enabled .Values.gateway.metricsServer.enabled (include "litellm.gateway.collectorSocketDir" .) }} + volumes: + {{- if .Values.gateway.config.create }} + - name: gateway-config + configMap: + name: {{ include "litellm.workload.fullname" . }}-config + {{- end }} + {{- if .Values.gateway.metricsServer.enabled }} + - name: prometheus-multiproc + emptyDir: {} + {{- end }} + {{- if include "litellm.gateway.collectorSocketDir" . }} + - name: collector-socket + emptyDir: + sizeLimit: 1Mi + {{- end }} + {{- if .Values.billingMetrics.enabled }} + {{- include "litellm.billingMetricsVolumes" . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.volumes }} + {{- toYaml . | nindent 8 }} + {{- end }} + {{- end }} + {{- with .Values.gateway.nodeSelector }} + nodeSelector: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.affinity }} + affinity: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.topologySpreadConstraints }} + topologySpreadConstraints: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- $gracePeriod := .Values.gateway.terminationGracePeriodSeconds }} + {{- if not (or (kindIs "invalid" $gracePeriod) (eq (printf "%v" $gracePeriod) "")) }} + terminationGracePeriodSeconds: {{ $gracePeriod }} + {{- end }} +{{- end -}} + +{{- define "litellm.workload.service" -}} +{{- $component := include "litellm.workload.componentName" . -}} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "litellm.workload.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} + {{- with .Values.gateway.service.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + type: {{ .Values.gateway.service.type }} + {{- with include "litellm.service.extras" .Values.gateway.service }} + {{- . | nindent 2 }} + {{- end }} + ports: + - port: {{ .Values.gateway.service.port }} + targetPort: http + protocol: TCP + name: http + selector: + {{- include "litellm.workload.selectorLabels" . | nindent 4 }} +{{- end -}} + +{{- define "litellm.workload.hpa" -}} +{{- $component := include "litellm.workload.componentName" . -}} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "litellm.workload.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: {{ include "litellm.workload.fullname" . }} + minReplicas: {{ .Values.gateway.hpa.minReplicas }} + maxReplicas: {{ .Values.gateway.hpa.maxReplicas }} + metrics: + {{- if .Values.gateway.hpa.targetCPUUtilizationPercentage }} + {{- if and .Values.gateway.collector.enabled .Values.gateway.collector.scaleOnGatewayContainerCpu }} + - type: ContainerResource + containerResource: + name: cpu + container: {{ $component }} + target: + type: Utilization + averageUtilization: {{ .Values.gateway.hpa.targetCPUUtilizationPercentage }} + {{- else }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ .Values.gateway.hpa.targetCPUUtilizationPercentage }} + {{- end }} + {{- end }} + {{- if .Values.gateway.hpa.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ .Values.gateway.hpa.targetMemoryUtilizationPercentage }} + {{- end }} + {{- with .Values.gateway.hpa.targetRequestsPerSecond }} + - type: Pods + pods: + metric: + name: litellm_requests_per_second + target: + type: AverageValue + averageValue: {{ toJson . | trimAll "\"" | quote }} + {{- end }} + {{- with .Values.gateway.hpa.targetTokensPerSecond }} + - type: Pods + pods: + metric: + name: litellm_tokens_per_second + target: + type: AverageValue + averageValue: {{ toJson . | trimAll "\"" | quote }} + {{- end }} + {{- with .Values.gateway.hpa.behavior }} + behavior: + {{- toYaml . | nindent 4 }} + {{- end }} +{{- end -}} + +{{/* +KEDA ScaledObject targeting the workload Deployment. The optional Prometheus +triggers query the same per pod counters the HPA workload metrics use, +scoped to this release's metrics Service through the `job` label the +ServiceMonitor gives every scrape. +*/}} +{{- define "litellm.workload.keda" -}} +{{- $component := include "litellm.workload.componentName" . -}} +{{- $keda := .Values.gateway.keda -}} +{{- $prom := $keda.prometheus -}} +{{- $wantsPrometheus := or $prom.targetRequestsPerSecond $prom.targetTokensPerSecond -}} +{{- if and $wantsPrometheus (not $prom.serverAddress) }} +{{- fail "gateway.keda.prometheus.serverAddress is required when gateway.keda.prometheus.targetRequestsPerSecond or targetTokensPerSecond is set" }} +{{- end }} +{{- $selector := printf "namespace=%q,job=%q" .Release.Namespace (printf "%s-metrics" (include "litellm.workload.fullname" .)) -}} +apiVersion: keda.sh/v1alpha1 +kind: ScaledObject +metadata: + name: {{ include "litellm.workload.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: {{ include "litellm.workload.fullname" . }} + minReplicaCount: {{ $keda.minReplicaCount }} + maxReplicaCount: {{ $keda.maxReplicaCount }} + pollingInterval: {{ $keda.pollingInterval }} + cooldownPeriod: {{ $keda.cooldownPeriod }} + {{- with $keda.advanced }} + advanced: + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with $keda.fallback }} + fallback: + {{- toYaml . | nindent 4 }} + {{- end }} + triggers: + {{- with $keda.triggers }} + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with $prom.targetRequestsPerSecond }} + - type: prometheus + metricType: AverageValue + metadata: + serverAddress: {{ $prom.serverAddress | quote }} + query: {{ printf "sum(rate(litellm_proxy_total_requests_metric_total{%s}[1m]))" $selector | quote }} + threshold: {{ toJson . | trimAll "\"" | quote }} + {{- end }} + {{- with $prom.targetTokensPerSecond }} + - type: prometheus + metricType: AverageValue + metadata: + serverAddress: {{ $prom.serverAddress | quote }} + query: {{ printf "sum(rate(litellm_total_tokens_metric_total{%s}[1m]))" $selector | quote }} + threshold: {{ toJson . | trimAll "\"" | quote }} + {{- end }} +{{- end -}} + +{{- define "litellm.workload.serviceMetrics" -}} +{{- $component := include "litellm.workload.componentName" . -}} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "litellm.workload.fullname" . }}-metrics + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} +spec: + type: ClusterIP + ports: + - port: {{ .Values.gateway.metricsServer.port }} + targetPort: metrics + protocol: TCP + name: metrics + selector: + {{- include "litellm.workload.selectorLabels" . | nindent 4 }} +{{- end -}} + +{{- define "litellm.workload.serviceMonitor" -}} +{{- $component := include "litellm.workload.componentName" . -}} +{{- if not .Values.gateway.metricsServer.enabled }} +{{- fail "gateway.serviceMonitor.enabled requires gateway.metricsServer.enabled: the http port serves /metrics/ behind virtual-key auth, so an unauthenticated scrape gets 401" }} +{{- end }} +apiVersion: monitoring.coreos.com/v1 +kind: ServiceMonitor +metadata: + name: {{ include "litellm.workload.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ $component }} + {{- with .Values.gateway.serviceMonitor.labels }} + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + selector: + matchLabels: + {{- include "litellm.workload.selectorLabels" . | nindent 6 }} + namespaceSelector: + matchNames: + - {{ .Release.Namespace | quote }} + endpoints: + - port: metrics + path: /metrics/ + interval: {{ .Values.gateway.serviceMonitor.interval }} + scrapeTimeout: {{ .Values.gateway.serviceMonitor.scrapeTimeout }} + scheme: http +{{- end -}} diff --git a/helm/litellm/templates/backend/deployment.yaml b/helm/litellm/templates/backend/deployment.yaml index 3eb64e5528c..066d7caaefc 100644 --- a/helm/litellm/templates/backend/deployment.yaml +++ b/helm/litellm/templates/backend/deployment.yaml @@ -1,4 +1,4 @@ -{{- if .Values.backend.enabled }} +{{- if include "litellm.backend.render" . }} apiVersion: apps/v1 kind: Deployment metadata: @@ -6,10 +6,21 @@ metadata: labels: {{- include "litellm.commonLabels" . | nindent 4 }} app.kubernetes.io/component: backend + {{- with .Values.backend.deploymentLabels }} + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with .Values.backend.deploymentAnnotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} spec: {{- if and (not .Values.backend.hpa.enabled) (not (kindIs "invalid" .Values.backend.replicaCount)) }} replicas: {{ .Values.backend.replicaCount }} {{- end }} + {{- $minReady := .Values.backend.minReadySeconds }} + {{- if not (or (kindIs "invalid" $minReady) (eq (printf "%v" $minReady) "")) }} + minReadySeconds: {{ $minReady }} + {{- end }} {{- with .Values.backend.strategy }} strategy: {{- toYaml . | nindent 4 }} @@ -44,19 +55,30 @@ spec: imagePullSecrets: {{- toYaml . | nindent 8 }} {{- end }} + {{- with .Values.backend.extraInitContainers }} + initContainers: + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} containers: - name: backend - image: "{{ .Values.backend.image.repository }}:{{ .Values.backend.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.backend.image.pullPolicy }} + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} {{- with .Values.backend.securityContext }} securityContext: {{- toYaml . | nindent 12 }} {{- end }} + args: + - backend + - --host + - 0.0.0.0 + - --port + - "4001" ports: - name: http containerPort: 4001 protocol: TCP env: + {{- include "litellm.masterKeyEnv" $ | nindent 12 }} {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.backend) | nindent 12 }} {{- if .Values.gateway.config.create }} - name: CONFIG_FILE_PATH @@ -106,7 +128,7 @@ spec: {{- if .Values.gateway.config.create }} - name: gateway-config configMap: - name: {{ include "litellm.gateway.fullname" . }}-config + name: {{ include "litellm.workload.fullname" . }}-config {{- end }} {{- if .Values.billingMetrics.enabled }} {{- include "litellm.billingMetricsVolumes" . | nindent 8 }} diff --git a/helm/litellm/templates/backend/hpa.yaml b/helm/litellm/templates/backend/hpa.yaml index a414092fb39..62efaacb830 100644 --- a/helm/litellm/templates/backend/hpa.yaml +++ b/helm/litellm/templates/backend/hpa.yaml @@ -1,4 +1,4 @@ -{{- if and .Values.backend.enabled .Values.backend.hpa.enabled }} +{{- if and (include "litellm.backend.render" .) .Values.backend.hpa.enabled }} apiVersion: autoscaling/v2 kind: HorizontalPodAutoscaler metadata: diff --git a/helm/litellm/templates/backend/poddisruptionbudget.yaml b/helm/litellm/templates/backend/poddisruptionbudget.yaml index 02853ac879c..ed40bda0d7c 100644 --- a/helm/litellm/templates/backend/poddisruptionbudget.yaml +++ b/helm/litellm/templates/backend/poddisruptionbudget.yaml @@ -1,5 +1,6 @@ {{- include "litellm.pdb" (dict "root" $ + "enabled" (include "litellm.backend.render" .) "component" .Values.backend "componentName" "backend" "fullname" (include "litellm.backend.fullname" .) diff --git a/helm/litellm/templates/backend/service.yaml b/helm/litellm/templates/backend/service.yaml index d480c654784..cd11ad95ccf 100644 --- a/helm/litellm/templates/backend/service.yaml +++ b/helm/litellm/templates/backend/service.yaml @@ -1,4 +1,4 @@ -{{- if .Values.backend.enabled }} +{{- if include "litellm.backend.render" . }} apiVersion: v1 kind: Service metadata: @@ -6,8 +6,15 @@ metadata: labels: {{- include "litellm.commonLabels" . | nindent 4 }} app.kubernetes.io/component: backend + {{- with .Values.backend.service.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} spec: type: {{ .Values.backend.service.type }} + {{- with include "litellm.service.extras" .Values.backend.service }} + {{- . | nindent 2 }} + {{- end }} ports: - port: {{ .Values.backend.service.port }} targetPort: http diff --git a/helm/litellm/templates/extra-resources.yaml b/helm/litellm/templates/extra-resources.yaml new file mode 100644 index 00000000000..bd809ffb696 --- /dev/null +++ b/helm/litellm/templates/extra-resources.yaml @@ -0,0 +1,4 @@ +{{- range .Values.extraResources }} +--- +{{ tpl (toYaml .) $ }} +{{- end }} diff --git a/helm/litellm/templates/gateway/configmap.yaml b/helm/litellm/templates/gateway/configmap.yaml index d262bf25b87..8944aaa0a2f 100644 --- a/helm/litellm/templates/gateway/configmap.yaml +++ b/helm/litellm/templates/gateway/configmap.yaml @@ -2,7 +2,10 @@ apiVersion: v1 kind: ConfigMap metadata: - name: {{ include "litellm.gateway.fullname" . }}-config + name: {{ include "litellm.workload.fullname" . }}-config + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: {{ include "litellm.workload.componentName" . }} data: config.yaml: | {{ .Values.gateway.config.proxy_config | toYaml | indent 6 }} diff --git a/helm/litellm/templates/gateway/deployment.yaml b/helm/litellm/templates/gateway/deployment.yaml index 49b452b3053..9db08850e26 100644 --- a/helm/litellm/templates/gateway/deployment.yaml +++ b/helm/litellm/templates/gateway/deployment.yaml @@ -1,247 +1,3 @@ -{{- if .Values.gateway.enabled }} -apiVersion: apps/v1 -kind: Deployment -metadata: - name: {{ include "litellm.gateway.fullname" . }} - labels: - {{- include "litellm.commonLabels" . | nindent 4 }} - app.kubernetes.io/component: gateway -spec: - {{- if and (not .Values.gateway.hpa.enabled) (not (kindIs "invalid" .Values.gateway.replicaCount)) }} - replicas: {{ .Values.gateway.replicaCount }} - {{- end }} - {{- with .Values.gateway.strategy }} - strategy: - {{- toYaml . | nindent 4 }} - {{- end }} - selector: - matchLabels: - {{- include "litellm.gateway.selectorLabels" . | nindent 6 }} - template: - metadata: - annotations: - {{- if .Values.gateway.config.create }} - checksum/config: {{ include (print $.Template.BasePath "/gateway/configmap.yaml") . | sha256sum }} - {{- end }} - {{- with .Values.gateway.podAnnotations }} - {{- toYaml . | nindent 8 }} - {{- end }} - labels: - {{- include "litellm.gateway.selectorLabels" . | nindent 8 }} - {{- with .Values.gateway.podLabels }} - {{- include "litellm.podLabels" (dict "podLabels" . "componentName" "gateway") | nindent 8 }} - {{- end }} - spec: - serviceAccountName: {{ include "litellm.gateway.serviceAccountName" . }} - automountServiceAccountToken: {{ .Values.serviceAccounts.gateway.automount }} - {{- with .Values.gateway.podSecurityContext }} - securityContext: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.imagePullSecrets }} - imagePullSecrets: - {{- toYaml . | nindent 8 }} - {{- end }} - containers: - - name: gateway - image: "{{ .Values.gateway.image.repository }}:{{ .Values.gateway.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.gateway.image.pullPolicy }} - {{- with .Values.gateway.securityContext }} - securityContext: - {{- toYaml . | nindent 12 }} - {{- end }} - ports: - - name: http - containerPort: 4000 - protocol: TCP - env: - {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.gateway) | nindent 12 }} - {{- if .Values.gateway.config.create }} - - name: CONFIG_FILE_PATH - value: /app/config/config.yaml - {{- end }} - {{- if .Values.gateway.numWorkers }} - - name: NUM_WORKERS - value: {{ .Values.gateway.numWorkers | quote }} - {{- end }} - {{- if .Values.database.connectionPool.enabled }} - {{- include "litellm.connectionPoolEnv" $ | nindent 12 }} - {{- end }} - {{- if .Values.billingMetrics.enabled }} - {{- include "litellm.billingMetricsEnv" . | nindent 12 }} - {{- end }} - {{- if .Values.gateway.metricsServer.enabled }} - {{- if eq (int .Values.gateway.metricsServer.port) 4000 }} - {{- fail "gateway.metricsServer.port must differ from the gateway port 4000" }} - {{- end }} - - name: PROMETHEUS_MULTIPROC_DIR - value: {{ include "litellm.gateway.prometheusMultiprocDir" . }} - {{- end }} - {{- if .Values.gateway.collector.enabled }} - {{- include "litellm.gateway.collectorEnv" . | nindent 12 }} - {{- end }} - {{- include "litellm.envFrom" .Values.gateway | nindent 10 }} - {{- if or .Values.gateway.config.create .Values.gateway.volumeMounts .Values.billingMetrics.enabled .Values.gateway.metricsServer.enabled (include "litellm.gateway.collectorSocketDir" .) }} - volumeMounts: - {{- if .Values.gateway.config.create }} - - name: gateway-config - mountPath: /app/config/config.yaml - subPath: config.yaml - {{- end }} - {{- if .Values.gateway.metricsServer.enabled }} - - name: prometheus-multiproc - mountPath: {{ include "litellm.gateway.prometheusMultiprocDir" . }} - {{- end }} - {{- if include "litellm.gateway.collectorSocketDir" . }} - - name: collector-socket - mountPath: {{ include "litellm.gateway.collectorSocketDir" . }} - {{- end }} - {{- if .Values.billingMetrics.enabled }} - {{- include "litellm.billingMetricsVolumeMounts" . | nindent 12 }} - {{- end }} - {{- with .Values.gateway.volumeMounts }} - {{- toYaml . | nindent 12 }} - {{- end }} - {{- end }} - {{- with .Values.gateway.livenessProbe }} - livenessProbe: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.gateway.readinessProbe }} - readinessProbe: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.gateway.startupProbe }} - startupProbe: - {{- toYaml . | nindent 12 }} - {{- end }} - {{- with .Values.gateway.lifecycle }} - lifecycle: - {{- toYaml . | nindent 12 }} - {{- end }} - resources: - {{- toYaml .Values.gateway.resources | nindent 12 }} - {{- if .Values.gateway.metricsServer.enabled }} - - name: metrics - image: "{{ .Values.gateway.image.repository }}:{{ .Values.gateway.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.gateway.image.pullPolicy }} - {{- with .Values.gateway.securityContext }} - securityContext: - {{- toYaml . | nindent 12 }} - {{- end }} - command: - - python - - -m - - litellm.proxy.prometheus_metrics_server - - --port - - {{ .Values.gateway.metricsServer.port | quote }} - env: - - name: PROMETHEUS_MULTIPROC_DIR - value: {{ include "litellm.gateway.prometheusMultiprocDir" . }} - ports: - - name: metrics - containerPort: {{ .Values.gateway.metricsServer.port }} - protocol: TCP - volumeMounts: - - name: prometheus-multiproc - mountPath: {{ include "litellm.gateway.prometheusMultiprocDir" . }} - readinessProbe: - tcpSocket: { port: metrics } - periodSeconds: 10 - livenessProbe: - tcpSocket: { port: metrics } - periodSeconds: 15 - failureThreshold: 6 - resources: - {{- toYaml .Values.gateway.metricsServer.resources | nindent 12 }} - {{- end }} - {{- if .Values.gateway.collector.enabled }} - - name: collector - image: "{{ .Values.gateway.image.repository }}:{{ .Values.gateway.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.gateway.image.pullPolicy }} - {{- with .Values.gateway.securityContext }} - securityContext: - {{- toYaml . | nindent 12 }} - {{- end }} - command: - - python - - -m - - litellm.proxy.collector - env: - {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.gateway) | nindent 12 }} - {{- if .Values.gateway.config.create }} - - name: CONFIG_FILE_PATH - value: /app/config/config.yaml - {{- end }} - {{- if .Values.database.connectionPool.enabled }} - {{- include "litellm.connectionPoolEnv" $ | nindent 12 }} - {{- end }} - {{- include "litellm.gateway.collectorEnv" . | nindent 12 }} - - name: LITELLM_JOB_ROLE - value: collector - {{- include "litellm.envFrom" .Values.gateway | nindent 10 }} - {{- if or .Values.gateway.config.create .Values.gateway.volumeMounts (include "litellm.gateway.collectorSocketDir" .) }} - volumeMounts: - {{- if .Values.gateway.config.create }} - - name: gateway-config - mountPath: /app/config/config.yaml - subPath: config.yaml - {{- end }} - {{- if include "litellm.gateway.collectorSocketDir" . }} - - name: collector-socket - mountPath: {{ include "litellm.gateway.collectorSocketDir" . }} - {{- end }} - {{- with .Values.gateway.volumeMounts }} - {{- toYaml . | nindent 12 }} - {{- end }} - {{- end }} - resources: - {{- toYaml .Values.gateway.collector.resources | nindent 12 }} - {{- end }} - {{- with .Values.gateway.extraContainers }} - {{- tpl (toYaml .) $ | nindent 8 }} - {{- end }} - {{- if or .Values.gateway.config.create .Values.gateway.volumes .Values.billingMetrics.enabled .Values.gateway.metricsServer.enabled (include "litellm.gateway.collectorSocketDir" .) }} - volumes: - {{- if .Values.gateway.config.create }} - - name: gateway-config - configMap: - name: {{ include "litellm.gateway.fullname" . }}-config - {{- end }} - {{- if .Values.gateway.metricsServer.enabled }} - - name: prometheus-multiproc - emptyDir: {} - {{- end }} - {{- if include "litellm.gateway.collectorSocketDir" . }} - - name: collector-socket - emptyDir: - sizeLimit: 1Mi - {{- end }} - {{- if .Values.billingMetrics.enabled }} - {{- include "litellm.billingMetricsVolumes" . | nindent 8 }} - {{- end }} - {{- with .Values.gateway.volumes }} - {{- toYaml . | nindent 8 }} - {{- end }} - {{- end }} - {{- with .Values.gateway.nodeSelector }} - nodeSelector: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.gateway.affinity }} - affinity: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.gateway.tolerations }} - tolerations: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- with .Values.gateway.topologySpreadConstraints }} - topologySpreadConstraints: - {{- toYaml . | nindent 8 }} - {{- end }} - {{- $gracePeriod := .Values.gateway.terminationGracePeriodSeconds }} - {{- if not (or (kindIs "invalid" $gracePeriod) (eq (printf "%v" $gracePeriod) "")) }} - terminationGracePeriodSeconds: {{ $gracePeriod }} - {{- end }} +{{- if include "litellm.gateway.render" . }} +{{- include "litellm.workload.deployment" . }} {{- end }} diff --git a/helm/litellm/templates/gateway/hpa.yaml b/helm/litellm/templates/gateway/hpa.yaml index e7094e96106..d416a6056bd 100644 --- a/helm/litellm/templates/gateway/hpa.yaml +++ b/helm/litellm/templates/gateway/hpa.yaml @@ -1,65 +1,3 @@ -{{- if and .Values.gateway.enabled .Values.gateway.hpa.enabled }} -apiVersion: autoscaling/v2 -kind: HorizontalPodAutoscaler -metadata: - name: {{ include "litellm.gateway.fullname" . }} - labels: - {{- include "litellm.commonLabels" . | nindent 4 }} - app.kubernetes.io/component: gateway -spec: - scaleTargetRef: - apiVersion: apps/v1 - kind: Deployment - name: {{ include "litellm.gateway.fullname" . }} - minReplicas: {{ .Values.gateway.hpa.minReplicas }} - maxReplicas: {{ .Values.gateway.hpa.maxReplicas }} - metrics: - {{- if .Values.gateway.hpa.targetCPUUtilizationPercentage }} - {{- if and .Values.gateway.collector.enabled .Values.gateway.collector.scaleOnGatewayContainerCpu }} - - type: ContainerResource - containerResource: - name: cpu - container: gateway - target: - type: Utilization - averageUtilization: {{ .Values.gateway.hpa.targetCPUUtilizationPercentage }} - {{- else }} - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: {{ .Values.gateway.hpa.targetCPUUtilizationPercentage }} - {{- end }} - {{- end }} - {{- if .Values.gateway.hpa.targetMemoryUtilizationPercentage }} - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: {{ .Values.gateway.hpa.targetMemoryUtilizationPercentage }} - {{- end }} - {{- with .Values.gateway.hpa.targetRequestsPerSecond }} - - type: Pods - pods: - metric: - name: litellm_requests_per_second - target: - type: AverageValue - averageValue: {{ toJson . | trimAll "\"" | quote }} - {{- end }} - {{- with .Values.gateway.hpa.targetTokensPerSecond }} - - type: Pods - pods: - metric: - name: litellm_tokens_per_second - target: - type: AverageValue - averageValue: {{ toJson . | trimAll "\"" | quote }} - {{- end }} - {{- with .Values.gateway.hpa.behavior }} - behavior: - {{- toYaml . | nindent 4 }} - {{- end }} +{{- if and (include "litellm.gateway.render" .) .Values.gateway.hpa.enabled }} +{{- include "litellm.workload.hpa" . }} {{- end }} diff --git a/helm/litellm/templates/gateway/keda.yaml b/helm/litellm/templates/gateway/keda.yaml new file mode 100644 index 00000000000..d46085ffef1 --- /dev/null +++ b/helm/litellm/templates/gateway/keda.yaml @@ -0,0 +1,3 @@ +{{- if and (include "litellm.gateway.render" .) .Values.gateway.keda.enabled }} +{{- include "litellm.workload.keda" . }} +{{- end }} diff --git a/helm/litellm/templates/gateway/poddisruptionbudget.yaml b/helm/litellm/templates/gateway/poddisruptionbudget.yaml index 15e89af17d7..313e3cc7535 100644 --- a/helm/litellm/templates/gateway/poddisruptionbudget.yaml +++ b/helm/litellm/templates/gateway/poddisruptionbudget.yaml @@ -1,5 +1,6 @@ {{- include "litellm.pdb" (dict "root" $ + "enabled" (include "litellm.gateway.render" .) "component" .Values.gateway "componentName" "gateway" "fullname" (include "litellm.gateway.fullname" .) diff --git a/helm/litellm/templates/gateway/service-metrics.yaml b/helm/litellm/templates/gateway/service-metrics.yaml index ad9bc05a9fd..e33ae8c63e3 100644 --- a/helm/litellm/templates/gateway/service-metrics.yaml +++ b/helm/litellm/templates/gateway/service-metrics.yaml @@ -1,18 +1,3 @@ -{{- if and .Values.gateway.enabled .Values.gateway.metricsServer.enabled }} -apiVersion: v1 -kind: Service -metadata: - name: {{ include "litellm.gateway.fullname" . }}-metrics - labels: - {{- include "litellm.commonLabels" . | nindent 4 }} - app.kubernetes.io/component: gateway -spec: - type: ClusterIP - ports: - - port: {{ .Values.gateway.metricsServer.port }} - targetPort: metrics - protocol: TCP - name: metrics - selector: - {{- include "litellm.gateway.selectorLabels" . | nindent 4 }} +{{- if and (include "litellm.gateway.render" .) .Values.gateway.metricsServer.enabled }} +{{- include "litellm.workload.serviceMetrics" . }} {{- end }} diff --git a/helm/litellm/templates/gateway/service.yaml b/helm/litellm/templates/gateway/service.yaml index 03a4167a0ab..8f3e7ad0673 100644 --- a/helm/litellm/templates/gateway/service.yaml +++ b/helm/litellm/templates/gateway/service.yaml @@ -1,18 +1,3 @@ -{{- if .Values.gateway.enabled }} -apiVersion: v1 -kind: Service -metadata: - name: {{ include "litellm.gateway.fullname" . }} - labels: - {{- include "litellm.commonLabels" . | nindent 4 }} - app.kubernetes.io/component: gateway -spec: - type: {{ .Values.gateway.service.type }} - ports: - - port: {{ .Values.gateway.service.port }} - targetPort: http - protocol: TCP - name: http - selector: - {{- include "litellm.gateway.selectorLabels" . | nindent 4 }} +{{- if include "litellm.gateway.render" . }} +{{- include "litellm.workload.service" . }} {{- end }} diff --git a/helm/litellm/templates/gateway/servicemonitor.yaml b/helm/litellm/templates/gateway/servicemonitor.yaml index e1bafa6e388..3e5d538b4e0 100644 --- a/helm/litellm/templates/gateway/servicemonitor.yaml +++ b/helm/litellm/templates/gateway/servicemonitor.yaml @@ -1,28 +1,3 @@ -{{- if and .Values.gateway.enabled .Values.gateway.serviceMonitor.enabled }} -{{- if not .Values.gateway.metricsServer.enabled }} -{{- fail "gateway.serviceMonitor.enabled requires gateway.metricsServer.enabled: the http port serves /metrics/ behind virtual-key auth, so an unauthenticated scrape gets 401" }} -{{- end }} -apiVersion: monitoring.coreos.com/v1 -kind: ServiceMonitor -metadata: - name: {{ include "litellm.gateway.fullname" . }} - labels: - {{- include "litellm.commonLabels" . | nindent 4 }} - app.kubernetes.io/component: gateway - {{- with .Values.gateway.serviceMonitor.labels }} - {{- toYaml . | nindent 4 }} - {{- end }} -spec: - selector: - matchLabels: - {{- include "litellm.gateway.selectorLabels" . | nindent 6 }} - namespaceSelector: - matchNames: - - {{ .Release.Namespace | quote }} - endpoints: - - port: metrics - path: /metrics/ - interval: {{ .Values.gateway.serviceMonitor.interval }} - scrapeTimeout: {{ .Values.gateway.serviceMonitor.scrapeTimeout }} - scheme: http +{{- if and (include "litellm.gateway.render" .) .Values.gateway.serviceMonitor.enabled }} +{{- include "litellm.workload.serviceMonitor" . }} {{- end }} diff --git a/helm/litellm/templates/ingress.yaml b/helm/litellm/templates/ingress.yaml index e9f7ed4ec3f..a86992bbeab 100644 --- a/helm/litellm/templates/ingress.yaml +++ b/helm/litellm/templates/ingress.yaml @@ -1,21 +1,26 @@ {{- if .Values.ingress.enabled -}} +{{- $monolith := .Values.monolith.enabled -}} {{- $gatewayName := include "litellm.gateway.fullname" . -}} {{- $backendName := include "litellm.backend.fullname" . -}} {{- $uiName := include "litellm.ui.fullname" . -}} {{- $gatewayPort := .Values.gateway.service.port -}} {{- $backendPort := .Values.backend.service.port -}} {{- $uiPort := .Values.ui.service.port -}} +{{- $proxyName := include "litellm.proxy.fullname" . -}} {{- $controller := .Values.ingress.controller | default "alb" -}} {{- if not (has $controller (list "alb" "nginx")) }} {{- fail (printf "ingress.controller: unknown controller %q, expected one of alb, nginx" $controller) }} {{- end }} {{/* Backends addressable from ingress.extraPaths, keyed by the `service` field. + In monolith mode the one proxy Service serves every component, so every + key resolves to it and an entry's `service` only has to be a known name. */}} +{{- $proxyBackend := dict "name" $proxyName "port" $gatewayPort -}} {{- $extraPathBackends := dict - "gateway" (dict "name" $gatewayName "port" $gatewayPort) - "backend" (dict "name" $backendName "port" $backendPort) - "ui" (dict "name" $uiName "port" $uiPort) + "gateway" (ternary $proxyBackend (dict "name" $gatewayName "port" $gatewayPort) $monolith) + "backend" (ternary $proxyBackend (dict "name" $backendName "port" $backendPort) $monolith) + "ui" (ternary $proxyBackend (dict "name" $uiName "port" $uiPort) $monolith) -}} {{/* UI paths (Next.js static export). @@ -89,7 +94,7 @@ at "/" Prefix would swallow the whole backend management API) instead of adding to it. */}} -{{- $builtinPathKeys := list "/test|Exact" "/debug/memory/summary|Exact" "/|Prefix" -}} +{{- $builtinPathKeys := ternary (list "/|Prefix") (list "/test|Exact" "/debug/memory/summary|Exact" "/|Prefix") $monolith -}} apiVersion: networking.k8s.io/v1 kind: Ingress metadata: @@ -114,6 +119,7 @@ spec: {{- end }} http: paths: + {{- if not $monolith }} # --- UI (Next.js static export) --- {{- range $uiPaths }} {{- $pathType := include "litellm.ingress.pathType" (dict "controller" $controller "path" .path "pathType" .pathType) }} @@ -156,6 +162,7 @@ spec: port: number: {{ $gatewayPort }} {{- end }} + {{- end }} {{- /* --- Operator-supplied extra paths (ingress.extraPaths) --- Rendered after every built-in path so an entry can never take @@ -198,6 +205,16 @@ spec: port: number: {{ $target.port }} {{- end }} + {{- if $monolith }} + # --- Monolith: the proxy serves the UI, the data plane and the management API --- + - path: / + pathType: Prefix + backend: + service: + name: {{ $proxyName }} + port: + number: {{ $gatewayPort }} + {{- else }} # --- Catch-all → backend (management API: /key/*, /user/*, /team/*, ...) --- - path: / pathType: Prefix @@ -206,4 +223,5 @@ spec: name: {{ $backendName }} port: number: {{ $backendPort }} + {{- end }} {{- end }} diff --git a/helm/litellm/templates/migrations-job.yaml b/helm/litellm/templates/migrations-job.yaml index de1cc2b103b..5b5738d726b 100644 --- a/helm/litellm/templates/migrations-job.yaml +++ b/helm/litellm/templates/migrations-job.yaml @@ -1,7 +1,7 @@ {{- if .Values.migrationJob.enabled -}} # Pre-install / pre-upgrade hook that runs `prisma migrate deploy` against -# the writer database before the gateway and backend Deployments are rolled -# out. Required because the gateway and backend both spin up Prisma at +# the writer database before the application Deployments are rolled out. +# Required because the proxy, gateway and backend all spin up Prisma at # startup and assume the LiteLLM schema (LiteLLM_Config, # LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...) already exists. # @@ -16,9 +16,10 @@ metadata: labels: {{- include "litellm.commonLabels" . | nindent 4 }} app.kubernetes.io/component: migrations - {{- if or .Values.migrationJob.hooks.helm.enabled .Values.migrationJob.hooks.argocd.enabled }} + {{- $helmHook := .Values.migrationJob.hooks.helm.enabled }} + {{- if or $helmHook .Values.migrationJob.hooks.argocd.enabled }} annotations: - {{- if .Values.migrationJob.hooks.helm.enabled }} + {{- if $helmHook }} helm.sh/hook: pre-install,pre-upgrade helm.sh/hook-delete-policy: before-hook-creation helm.sh/hook-weight: {{ .Values.migrationJob.hooks.helm.weight | default "0" | quote }} @@ -55,14 +56,20 @@ spec: imagePullSecrets: {{- toYaml . | nindent 8 }} {{- end }} + {{- with .Values.migrationJob.extraInitContainers }} + initContainers: + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} containers: - name: prisma-migrations - image: "{{ .Values.migrationJob.image.repository }}:{{ .Values.migrationJob.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.migrationJob.image.pullPolicy }} + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} {{- with .Values.migrationJob.securityContext }} securityContext: {{- toYaml . | nindent 12 }} {{- end }} + args: + - migrations env: {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.migrationJob) | nindent 12 }} {{- with .Values.migrationJob.volumeMounts }} @@ -73,6 +80,9 @@ spec: resources: {{- toYaml . | nindent 12 }} {{- end }} + {{- with .Values.migrationJob.extraContainers }} + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} {{- with .Values.migrationJob.volumes }} volumes: {{- toYaml . | nindent 8 }} diff --git a/helm/litellm/templates/monolith/deployment.yaml b/helm/litellm/templates/monolith/deployment.yaml new file mode 100644 index 00000000000..7bad16386e1 --- /dev/null +++ b/helm/litellm/templates/monolith/deployment.yaml @@ -0,0 +1,3 @@ +{{- if .Values.monolith.enabled }} +{{- include "litellm.workload.deployment" . }} +{{- end }} diff --git a/helm/litellm/templates/monolith/hpa.yaml b/helm/litellm/templates/monolith/hpa.yaml new file mode 100644 index 00000000000..ee9180e71bb --- /dev/null +++ b/helm/litellm/templates/monolith/hpa.yaml @@ -0,0 +1,3 @@ +{{- if and .Values.monolith.enabled .Values.gateway.hpa.enabled }} +{{- include "litellm.workload.hpa" . }} +{{- end }} diff --git a/helm/litellm/templates/monolith/keda.yaml b/helm/litellm/templates/monolith/keda.yaml new file mode 100644 index 00000000000..d4054432e38 --- /dev/null +++ b/helm/litellm/templates/monolith/keda.yaml @@ -0,0 +1,3 @@ +{{- if and .Values.monolith.enabled .Values.gateway.keda.enabled }} +{{- include "litellm.workload.keda" . }} +{{- end }} diff --git a/helm/litellm/templates/monolith/poddisruptionbudget.yaml b/helm/litellm/templates/monolith/poddisruptionbudget.yaml new file mode 100644 index 00000000000..2be83e57dfc --- /dev/null +++ b/helm/litellm/templates/monolith/poddisruptionbudget.yaml @@ -0,0 +1,7 @@ +{{- include "litellm.pdb" (dict + "root" $ + "enabled" .Values.monolith.enabled + "component" .Values.gateway + "componentName" "proxy" + "fullname" (include "litellm.proxy.fullname" .) + "selectorLabels" (include "litellm.proxy.selectorLabels" .)) }} diff --git a/helm/litellm/templates/monolith/service-metrics.yaml b/helm/litellm/templates/monolith/service-metrics.yaml new file mode 100644 index 00000000000..07a68c453df --- /dev/null +++ b/helm/litellm/templates/monolith/service-metrics.yaml @@ -0,0 +1,3 @@ +{{- if and .Values.monolith.enabled .Values.gateway.metricsServer.enabled }} +{{- include "litellm.workload.serviceMetrics" . }} +{{- end }} diff --git a/helm/litellm/templates/monolith/service.yaml b/helm/litellm/templates/monolith/service.yaml new file mode 100644 index 00000000000..4f87259983a --- /dev/null +++ b/helm/litellm/templates/monolith/service.yaml @@ -0,0 +1,3 @@ +{{- if .Values.monolith.enabled }} +{{- include "litellm.workload.service" . }} +{{- end }} diff --git a/helm/litellm/templates/monolith/servicemonitor.yaml b/helm/litellm/templates/monolith/servicemonitor.yaml new file mode 100644 index 00000000000..29a51a379f2 --- /dev/null +++ b/helm/litellm/templates/monolith/servicemonitor.yaml @@ -0,0 +1,3 @@ +{{- if and .Values.monolith.enabled .Values.gateway.serviceMonitor.enabled }} +{{- include "litellm.workload.serviceMonitor" . }} +{{- end }} diff --git a/helm/litellm/templates/secret-masterkey.yaml b/helm/litellm/templates/secret-masterkey.yaml new file mode 100644 index 00000000000..3af2ee6f07d --- /dev/null +++ b/helm/litellm/templates/secret-masterkey.yaml @@ -0,0 +1,22 @@ +{{- if and .Values.masterKey.generate (not .Values.masterKey.secretName) }} +{{- $name := include "litellm.masterKey.generatedSecretName" . }} +{{- $key := .Values.masterKey.secretKey | default "master-key" }} +{{- $existing := lookup "v1" "Secret" .Release.Namespace $name }} +{{- $value := "" }} +{{- if and $existing (index $existing.data $key) }} +{{- $value = index $existing.data $key | b64dec }} +{{- else }} +{{- $value = printf "sk-%s" (randAlphaNum 32) }} +{{- end }} +apiVersion: v1 +kind: Secret +metadata: + name: {{ $name }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + annotations: + helm.sh/resource-policy: keep +type: Opaque +data: + {{ $key }}: {{ $value | b64enc }} +{{- end }} diff --git a/helm/litellm/templates/serviceaccount.yaml b/helm/litellm/templates/serviceaccount.yaml index a2fc52f47c0..9b1171a3e90 100644 --- a/helm/litellm/templates/serviceaccount.yaml +++ b/helm/litellm/templates/serviceaccount.yaml @@ -1,5 +1,5 @@ {{- $prev := false -}} -{{- if .Values.serviceAccounts.gateway.create -}} +{{- if and .Values.serviceAccounts.gateway.create (include "litellm.workload.render" .) -}} {{- $prev = true }} apiVersion: v1 kind: ServiceAccount @@ -14,7 +14,7 @@ metadata: {{- end }} automountServiceAccountToken: {{ .Values.serviceAccounts.gateway.automount }} {{- end }} -{{- if .Values.serviceAccounts.backend.create }} +{{- if and .Values.serviceAccounts.backend.create (include "litellm.backend.render" .) }} {{- if $prev }} --- {{- end }} @@ -32,7 +32,7 @@ metadata: {{- end }} automountServiceAccountToken: {{ .Values.serviceAccounts.backend.automount }} {{- end }} -{{- if .Values.serviceAccounts.ui.create }} +{{- if and .Values.serviceAccounts.ui.create (include "litellm.ui.render" .) }} {{- if $prev }} --- {{- end }} diff --git a/helm/litellm/templates/tests/test-connection.yaml b/helm/litellm/templates/tests/test-connection.yaml new file mode 100644 index 00000000000..bc258c92790 --- /dev/null +++ b/helm/litellm/templates/tests/test-connection.yaml @@ -0,0 +1,53 @@ +{{- if include "litellm.workload.render" . }} +apiVersion: v1 +kind: Pod +metadata: + name: {{ include "litellm.fullname" . }}-test-connection + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: test + annotations: + helm.sh/hook: test + helm.sh/hook-delete-policy: before-hook-creation +spec: + restartPolicy: Never + {{- with .Values.gateway.podSecurityContext }} + securityContext: + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 4 }} + {{- end }} + containers: + - name: health + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} + {{- with .Values.gateway.securityContext }} + securityContext: + {{- toYaml . | nindent 8 }} + {{- end }} + env: + - name: LITELLM_READINESS_URL + value: http://{{ include "litellm.workload.fullname" . }}:{{ .Values.gateway.service.port }}/health/readiness + args: + - python + - -c + - | + import json, os, sys, time, urllib.error, urllib.request + url = os.environ["LITELLM_READINESS_URL"] + for attempt in range(1, 31): + try: + with urllib.request.urlopen(url, timeout=5) as response: + body = response.read().decode() + print(url, response.status, body) + db = json.loads(body).get("db") + if db != "connected": + print(f"readiness reports db={db!r}: the proxy is not using the configured database") + sys.exit(1) + sys.exit(0) + except (urllib.error.URLError, OSError) as error: + print(f"attempt {attempt}: {url}: {error}") + time.sleep(2) + sys.exit(1) +{{- end }} diff --git a/helm/litellm/templates/ui/deployment.yaml b/helm/litellm/templates/ui/deployment.yaml index efee2d5fc34..94e85ef5266 100644 --- a/helm/litellm/templates/ui/deployment.yaml +++ b/helm/litellm/templates/ui/deployment.yaml @@ -1,4 +1,4 @@ -{{- if .Values.ui.enabled }} +{{- if include "litellm.ui.render" . }} apiVersion: apps/v1 kind: Deployment metadata: @@ -6,10 +6,21 @@ metadata: labels: {{- include "litellm.commonLabels" . | nindent 4 }} app.kubernetes.io/component: ui + {{- with .Values.ui.deploymentLabels }} + {{- toYaml . | nindent 4 }} + {{- end }} + {{- with .Values.ui.deploymentAnnotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} spec: {{- if and (not .Values.ui.hpa.enabled) (not (kindIs "invalid" .Values.ui.replicaCount)) }} replicas: {{ .Values.ui.replicaCount }} {{- end }} + {{- $minReady := .Values.ui.minReadySeconds }} + {{- if not (or (kindIs "invalid" $minReady) (eq (printf "%v" $minReady) "")) }} + minReadySeconds: {{ $minReady }} + {{- end }} {{- with .Values.ui.strategy }} strategy: {{- toYaml . | nindent 4 }} @@ -39,14 +50,20 @@ spec: imagePullSecrets: {{- toYaml . | nindent 8 }} {{- end }} + {{- with .Values.ui.extraInitContainers }} + initContainers: + {{- tpl (toYaml .) $ | nindent 8 }} + {{- end }} containers: - name: ui - image: "{{ .Values.ui.image.repository }}:{{ .Values.ui.image.tag | default .Chart.AppVersion }}" - imagePullPolicy: {{ .Values.ui.image.pullPolicy }} + image: {{ include "litellm.image" . | quote }} + imagePullPolicy: {{ .Values.image.pullPolicy }} {{- with .Values.ui.securityContext }} securityContext: {{- toYaml . | nindent 12 }} {{- end }} + args: + - ui ports: - name: http containerPort: 3000 diff --git a/helm/litellm/templates/ui/hpa.yaml b/helm/litellm/templates/ui/hpa.yaml index a9b0b51129e..67e17b269c7 100644 --- a/helm/litellm/templates/ui/hpa.yaml +++ b/helm/litellm/templates/ui/hpa.yaml @@ -1,4 +1,4 @@ -{{- if and .Values.ui.enabled .Values.ui.hpa.enabled }} +{{- if and (include "litellm.ui.render" .) .Values.ui.hpa.enabled }} apiVersion: autoscaling/v2 kind: HorizontalPodAutoscaler metadata: diff --git a/helm/litellm/templates/ui/poddisruptionbudget.yaml b/helm/litellm/templates/ui/poddisruptionbudget.yaml index f7a3a694e9c..fa4bb44612c 100644 --- a/helm/litellm/templates/ui/poddisruptionbudget.yaml +++ b/helm/litellm/templates/ui/poddisruptionbudget.yaml @@ -1,5 +1,6 @@ {{- include "litellm.pdb" (dict "root" $ + "enabled" (include "litellm.ui.render" .) "component" .Values.ui "componentName" "ui" "fullname" (include "litellm.ui.fullname" .) diff --git a/helm/litellm/templates/ui/service.yaml b/helm/litellm/templates/ui/service.yaml index 52b539fa00c..d544d9d6148 100644 --- a/helm/litellm/templates/ui/service.yaml +++ b/helm/litellm/templates/ui/service.yaml @@ -1,4 +1,4 @@ -{{- if .Values.ui.enabled }} +{{- if include "litellm.ui.render" . }} apiVersion: v1 kind: Service metadata: @@ -6,8 +6,15 @@ metadata: labels: {{- include "litellm.commonLabels" . | nindent 4 }} app.kubernetes.io/component: ui + {{- with .Values.ui.service.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} spec: type: {{ .Values.ui.service.type }} + {{- with include "litellm.service.extras" .Values.ui.service }} + {{- . | nindent 2 }} + {{- end }} ports: - port: {{ .Values.ui.service.port }} targetPort: http diff --git a/helm/litellm/tests/collector_tests.yaml b/helm/litellm/tests/collector_tests.yaml index 4ef7e3c8ca4..8e9804dece8 100644 --- a/helm/litellm/tests/collector_tests.yaml +++ b/helm/litellm/tests/collector_tests.yaml @@ -34,7 +34,7 @@ tests: gateway.collector.enabled: true gateway.collector.bufferSize: 250 gateway.collector.onUnavailable: drop - gateway.image.tag: v1.102.0 + image.tag: v1.102.0 gateway.numWorkers: 4 database.connectionPool.enabled: true database.connectionPool.maxDbConnections: 8 @@ -80,14 +80,12 @@ tests: template: gateway/deployment.yaml - equal: path: spec.template.spec.containers[1].image - value: ghcr.io/berriai/litellm-gateway:v1.102.0 + value: ghcr.io/berriai/litellm:v1.102.0 template: gateway/deployment.yaml - equal: - path: spec.template.spec.containers[1].command + path: spec.template.spec.containers[1].args value: - - python - - -m - - litellm.proxy.collector + - collector template: gateway/deployment.yaml - contains: path: spec.template.spec.containers[1].env diff --git a/helm/litellm/tests/extra_resources_tests.yaml b/helm/litellm/tests/extra_resources_tests.yaml new file mode 100644 index 00000000000..4935b3a5401 --- /dev/null +++ b/helm/litellm/tests/extra_resources_tests.yaml @@ -0,0 +1,45 @@ +suite: test extraResources +templates: + - extra-resources.yaml +values: + - ./values/required.yaml +tests: + - it: renders nothing by default + asserts: + - hasDocuments: + count: 0 + + - it: renders every entry as its own document and templates it against the release + set: + extraResources: + - apiVersion: v1 + kind: ConfigMap + metadata: + name: "{{ .Release.Name }}-extra" + data: + chart: "{{ .Chart.Name }}" + - apiVersion: networking.k8s.io/v1 + kind: NetworkPolicy + metadata: + name: deny-all + spec: + podSelector: {} + policyTypes: + - Ingress + asserts: + - hasDocuments: + count: 2 + - isKind: + of: ConfigMap + documentIndex: 0 + - equal: + path: metadata.name + value: RELEASE-NAME-extra + documentIndex: 0 + - equal: + path: data.chart + value: litellm + documentIndex: 0 + - isKind: + of: NetworkPolicy + documentIndex: 1 diff --git a/helm/litellm/tests/ingress_monolith_tests.yaml b/helm/litellm/tests/ingress_monolith_tests.yaml new file mode 100644 index 00000000000..be149753ce4 --- /dev/null +++ b/helm/litellm/tests/ingress_monolith_tests.yaml @@ -0,0 +1,130 @@ +suite: test ingress routing in monolith mode +templates: + - ingress.yaml +values: + - ./values/required.yaml +tests: + - it: componentized mode still splits traffic across ui, gateway and backend Services + set: + ingress.enabled: true + asserts: + - equal: + path: spec.rules[0].http.paths[0].backend.service.name + value: RELEASE-NAME-litellm-ui + - contains: + path: spec.rules[0].http.paths + content: + path: /v1/chat + pathType: Prefix + backend: + service: + name: RELEASE-NAME-litellm-gateway + port: + number: 4000 + - equal: + path: spec.rules[0].http.paths[-1].backend.service.name + value: RELEASE-NAME-litellm-backend + + - it: monolith mode routes a single catch-all to the monolith Service and nothing else + set: + monolith.enabled: true + ingress.enabled: true + asserts: + - lengthEqual: + path: spec.rules[0].http.paths + count: 1 + - equal: + path: spec.rules[0].http.paths[0] + value: + path: / + pathType: Prefix + backend: + service: + name: RELEASE-NAME-litellm + port: + number: 4000 + - notMatchRegexRaw: + pattern: RELEASE-NAME-litellm-(gateway|backend|ui) + + - it: monolith mode uses the gateway service port for the catch-all + set: + monolith.enabled: true + ingress.enabled: true + gateway.service.port: 8080 + asserts: + - equal: + path: spec.rules[0].http.paths[0].backend.service.port.number + value: 8080 + + - it: monolith mode sends every extraPaths entry to the monolith Service regardless of its service field + set: + monolith.enabled: true + ingress.enabled: true + ingress.extraPaths: + - path: /watsonx + - path: /admin + service: backend + pathType: Exact + - path: /dashboard + service: ui + asserts: + - lengthEqual: + path: spec.rules[0].http.paths + count: 4 + - equal: + path: spec.rules[0].http.paths[0] + value: + path: /watsonx + pathType: Prefix + backend: + service: + name: RELEASE-NAME-litellm + port: + number: 4000 + - equal: + path: spec.rules[0].http.paths[1].backend.service.name + value: RELEASE-NAME-litellm + - equal: + path: spec.rules[0].http.paths[1].pathType + value: Exact + - equal: + path: spec.rules[0].http.paths[2].backend.service.name + value: RELEASE-NAME-litellm + - equal: + path: spec.rules[0].http.paths[3].path + value: / + + - it: monolith mode keeps rejecting an extra path at the root + set: + monolith.enabled: true + ingress.enabled: true + ingress.extraPaths: + - path: / + asserts: + - failedTemplate: + errorPattern: "ingress.extraPaths\\[0\\]: path / is already routed" + + - it: monolith mode keeps the host, TLS and class settings + set: + monolith.enabled: true + ingress.enabled: true + ingress.className: nginx + ingress.controller: nginx + ingress.host: llm.example.com + ingress.tls: + - hosts: + - llm.example.com + secretName: llm-tls + asserts: + - equal: + path: spec.ingressClassName + value: nginx + - equal: + path: spec.rules[0].host + value: llm.example.com + - equal: + path: spec.tls[0].secretName + value: llm-tls + - equal: + path: spec.rules[0].http.paths[0].backend.service.name + value: RELEASE-NAME-litellm diff --git a/helm/litellm/tests/keda_tests.yaml b/helm/litellm/tests/keda_tests.yaml new file mode 100644 index 00000000000..dac782f94f8 --- /dev/null +++ b/helm/litellm/tests/keda_tests.yaml @@ -0,0 +1,201 @@ +suite: test KEDA ScaledObject +templates: + - gateway/keda.yaml + - gateway/hpa.yaml + - gateway/deployment.yaml + - gateway/configmap.yaml +values: + - ./values/required.yaml +release: + name: rel + namespace: llm +tests: + - it: renders no ScaledObject by default + template: gateway/keda.yaml + asserts: + - hasDocuments: + count: 0 + + - it: passes user triggers through and adds no prometheus triggers by default + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.triggers: + - type: cpu + metricType: Utilization + metadata: + value: "60" + templates: + - gateway/keda.yaml + - gateway/hpa.yaml + asserts: + - hasDocuments: + count: 0 + template: gateway/hpa.yaml + - isKind: + of: ScaledObject + template: gateway/keda.yaml + - equal: + path: metadata.name + value: rel-litellm-gateway + template: gateway/keda.yaml + - equal: + path: spec.scaleTargetRef + value: + apiVersion: apps/v1 + kind: Deployment + name: rel-litellm-gateway + template: gateway/keda.yaml + - equal: + path: spec.minReplicaCount + value: 1 + template: gateway/keda.yaml + - equal: + path: spec.maxReplicaCount + value: 10 + template: gateway/keda.yaml + - equal: + path: spec.triggers + value: + - type: cpu + metricType: Utilization + metadata: + value: "60" + template: gateway/keda.yaml + + - it: leaves spec.replicas off the Deployment so KEDA owns the replica count + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.replicaCount: 3 + template: gateway/deployment.yaml + asserts: + - notExists: + path: spec.replicas + + - it: scales on requests per second against the metrics Service job + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.prometheus.serverAddress: http://prometheus-operated.monitoring.svc:9090 + gateway.keda.prometheus.targetRequestsPerSecond: 90 + template: gateway/keda.yaml + asserts: + - lengthEqual: + path: spec.triggers + count: 1 + - equal: + path: spec.triggers[0] + value: + type: prometheus + metricType: AverageValue + metadata: + serverAddress: http://prometheus-operated.monitoring.svc:9090 + threshold: "90" + query: sum(rate(litellm_proxy_total_requests_metric_total{namespace="llm",job="rel-litellm-gateway-metrics"}[1m])) + + - it: scales on tokens per second on its own + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.prometheus.serverAddress: http://prom:9090 + gateway.keda.prometheus.targetTokensPerSecond: 6000000 + template: gateway/keda.yaml + asserts: + - lengthEqual: + path: spec.triggers + count: 1 + - equal: + path: spec.triggers[0].type + value: prometheus + - equal: + path: spec.triggers[0].metadata.threshold + value: "6000000" + - equal: + path: spec.triggers[0].metadata.query + value: sum(rate(litellm_total_tokens_metric_total{namespace="llm",job="rel-litellm-gateway-metrics"}[1m])) + + - it: appends the requests and tokens triggers after user triggers + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.metricsServer.enabled: true + gateway.keda.triggers: + - type: cpu + metricType: Utilization + metadata: + value: "60" + gateway.keda.prometheus.serverAddress: http://prom:9090 + gateway.keda.prometheus.targetRequestsPerSecond: 90 + gateway.keda.prometheus.targetTokensPerSecond: 6000000 + template: gateway/keda.yaml + asserts: + - lengthEqual: + path: spec.triggers + count: 3 + - equal: + path: spec.triggers[0].type + value: cpu + - equal: + path: spec.triggers[1].metadata.threshold + value: "90" + - equal: + path: spec.triggers[2].metadata.threshold + value: "6000000" + - notMatchRegexRaw: + pattern: "\\* *60|per_minute|PerMinute" + + - it: renders advanced, fallback, polling and cooldown settings + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.minReplicaCount: 2 + gateway.keda.maxReplicaCount: 20 + gateway.keda.pollingInterval: 15 + gateway.keda.cooldownPeriod: 120 + gateway.keda.fallback: + failureThreshold: 3 + replicas: 4 + gateway.keda.advanced: + restoreToOriginalReplicaCount: true + template: gateway/keda.yaml + asserts: + - equal: + path: spec.minReplicaCount + value: 2 + - equal: + path: spec.maxReplicaCount + value: 20 + - equal: + path: spec.pollingInterval + value: 15 + - equal: + path: spec.cooldownPeriod + value: 120 + - equal: + path: spec.fallback + value: + failureThreshold: 3 + replicas: 4 + - equal: + path: spec.advanced.restoreToOriginalReplicaCount + value: true + + - it: refuses a workload target without a prometheus server address + set: + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.prometheus.targetRequestsPerSecond: 90 + template: gateway/keda.yaml + asserts: + - failedTemplate: + errorMessage: gateway.keda.prometheus.serverAddress is required when gateway.keda.prometheus.targetRequestsPerSecond or targetTokensPerSecond is set + + - it: refuses to run the HPA and KEDA against the same Deployment + set: + gateway.hpa.enabled: true + gateway.keda.enabled: true + template: gateway/deployment.yaml + asserts: + - failedTemplate: + errorPattern: gateway.hpa.enabled and gateway.keda.enabled are mutually exclusive diff --git a/helm/litellm/tests/masterkey_secret_tests.yaml b/helm/litellm/tests/masterkey_secret_tests.yaml new file mode 100644 index 00000000000..89025e2ad48 --- /dev/null +++ b/helm/litellm/tests/masterkey_secret_tests.yaml @@ -0,0 +1,127 @@ +suite: test the generated master key Secret +templates: + - secret-masterkey.yaml + - gateway/deployment.yaml + - gateway/configmap.yaml + - monolith/deployment.yaml + - migrations-job.yaml +values: + - ./values/required.yaml +tests: + - it: renders no Secret when the operator supplies masterKey.secretName + template: secret-masterkey.yaml + asserts: + - hasDocuments: + count: 0 + + - it: refuses to render the workload without a Secret name when masterKey.generate is off + set: + masterKey.secretName: "" + template: gateway/deployment.yaml + asserts: + - failedTemplate: + errorPattern: masterKey.secretName is required + + - it: generates a Secret whose key starts with sk- (base64 c2st) and keeps it across uninstall + set: + masterKey.secretName: "" + masterKey.generate: true + template: secret-masterkey.yaml + asserts: + - isKind: + of: Secret + - equal: + path: metadata.name + value: RELEASE-NAME-litellm-masterkey + - equal: + path: metadata.annotations["helm.sh/resource-policy"] + value: keep + - matchRegex: + path: data["master-key"] + pattern: ^c2st + + - it: stores the value under masterKey.secretKey + set: + masterKey.secretName: "" + masterKey.generate: true + masterKey.secretKey: LITELLM_MASTER_KEY + template: secret-masterkey.yaml + asserts: + - matchRegex: + path: data.LITELLM_MASTER_KEY + pattern: ^c2st + + - it: reuses the master key already stored in the cluster instead of generating a new one on upgrade + set: + masterKey.secretName: "" + masterKey.generate: true + template: secret-masterkey.yaml + kubernetesProvider: + scheme: + "v1/Secret": + gvr: + version: "v1" + resource: "secrets" + namespaced: true + objects: + - kind: Secret + apiVersion: v1 + metadata: + name: RELEASE-NAME-litellm-masterkey + namespace: NAMESPACE + data: + master-key: c2stZXhpc3Rpbmcta2V5 + asserts: + - equal: + path: data["master-key"] + value: c2stZXhpc3Rpbmcta2V5 + + - it: points the app containers at the generated Secret and keeps it out of the pre-install migrations Job + set: + masterKey.secretName: "" + masterKey.generate: true + monolith.enabled: true + templates: + - monolith/deployment.yaml + - migrations-job.yaml + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: LITELLM_MASTER_KEY + valueFrom: + secretKeyRef: + name: RELEASE-NAME-litellm-masterkey + key: master-key + template: monolith/deployment.yaml + - equal: + path: metadata.annotations["helm.sh/hook"] + value: pre-install,pre-upgrade + template: migrations-job.yaml + - notContains: + path: spec.template.spec.containers[0].env + content: + name: LITELLM_MASTER_KEY + any: true + template: migrations-job.yaml + + - it: an explicit masterKey.secretName wins over masterKey.generate + set: + masterKey.secretName: operator-secret + masterKey.generate: true + templates: + - secret-masterkey.yaml + - gateway/deployment.yaml + asserts: + - hasDocuments: + count: 0 + template: secret-masterkey.yaml + - contains: + path: spec.template.spec.containers[0].env + content: + name: LITELLM_MASTER_KEY + valueFrom: + secretKeyRef: + name: operator-secret + key: master-key + template: gateway/deployment.yaml diff --git a/helm/litellm/tests/metrics_server_tests.yaml b/helm/litellm/tests/metrics_server_tests.yaml index 0e7d9d9e9ee..1a48b7be671 100644 --- a/helm/litellm/tests/metrics_server_tests.yaml +++ b/helm/litellm/tests/metrics_server_tests.yaml @@ -38,7 +38,7 @@ tests: gateway.metricsServer.enabled: true gateway.metricsServer.port: 4101 gateway.service.type: LoadBalancer - gateway.image.tag: v1.101.0 + image.tag: v1.101.0 asserts: - contains: path: spec.template.spec.containers[0].env @@ -58,14 +58,12 @@ tests: template: gateway/deployment.yaml - equal: path: spec.template.spec.containers[1].image - value: ghcr.io/berriai/litellm-gateway:v1.101.0 + value: ghcr.io/berriai/litellm:v1.101.0 template: gateway/deployment.yaml - equal: - path: spec.template.spec.containers[1].command + path: spec.template.spec.containers[1].args value: - - python - - -m - - litellm.proxy.prometheus_metrics_server + - metrics - --port - "4101" template: gateway/deployment.yaml diff --git a/helm/litellm/tests/migration_job_tests.yaml b/helm/litellm/tests/migration_job_tests.yaml index 2ebb1b44926..a62d5aaf243 100644 --- a/helm/litellm/tests/migration_job_tests.yaml +++ b/helm/litellm/tests/migration_job_tests.yaml @@ -91,7 +91,7 @@ tests: app.kubernetes.io/name: litellm app.kubernetes.io/instance: RELEASE-NAME app.kubernetes.io/managed-by: Helm - helm.sh/chart: litellm-0.1.0 + helm.sh/chart: litellm-1.0.0 app.kubernetes.io/component: migrations - it: renders pod-level and container-level securityContext in their own scopes diff --git a/helm/litellm/tests/monolith_tests.yaml b/helm/litellm/tests/monolith_tests.yaml new file mode 100644 index 00000000000..2c5274db43f --- /dev/null +++ b/helm/litellm/tests/monolith_tests.yaml @@ -0,0 +1,526 @@ +suite: test monolith mode +templates: + - monolith/deployment.yaml + - monolith/service.yaml + - monolith/hpa.yaml + - monolith/keda.yaml + - monolith/poddisruptionbudget.yaml + - monolith/service-metrics.yaml + - monolith/servicemonitor.yaml + - gateway/deployment.yaml + - gateway/service.yaml + - gateway/hpa.yaml + - gateway/keda.yaml + - gateway/poddisruptionbudget.yaml + - gateway/service-metrics.yaml + - gateway/servicemonitor.yaml + - gateway/configmap.yaml + - backend/deployment.yaml + - backend/service.yaml + - backend/hpa.yaml + - backend/poddisruptionbudget.yaml + - ui/deployment.yaml + - ui/service.yaml + - ui/hpa.yaml + - ui/poddisruptionbudget.yaml + - serviceaccount.yaml + - migrations-job.yaml +values: + - ./values/required.yaml +tests: + - it: renders no monolith resources by default + templates: + - monolith/deployment.yaml + - monolith/service.yaml + - monolith/hpa.yaml + - monolith/poddisruptionbudget.yaml + - monolith/service-metrics.yaml + - monolith/servicemonitor.yaml + asserts: + - hasDocuments: + count: 0 + + - it: componentized mode renders one Deployment per component + templates: + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + asserts: + - hasDocuments: + count: 1 + - isKind: + of: Deployment + + - it: monolith mode renders exactly one Deployment, running the proxy dispatcher + set: + monolith.enabled: true + templates: + - monolith/deployment.yaml + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + asserts: + - hasDocuments: + count: 1 + template: monolith/deployment.yaml + - hasDocuments: + count: 0 + template: gateway/deployment.yaml + - hasDocuments: + count: 0 + template: backend/deployment.yaml + - hasDocuments: + count: 0 + template: ui/deployment.yaml + - isKind: + of: Deployment + template: monolith/deployment.yaml + - equal: + path: metadata.name + value: RELEASE-NAME-litellm + template: monolith/deployment.yaml + - equal: + path: metadata.labels["app.kubernetes.io/component"] + value: proxy + template: monolith/deployment.yaml + - equal: + path: spec.template.spec.containers[0].name + value: proxy + template: monolith/deployment.yaml + - equal: + path: spec.template.spec.containers[0].args[0] + value: proxy + template: monolith/deployment.yaml + - equal: + path: spec.template.spec.containers[0].args + value: + - proxy + - --port + - "4000" + - --config + - /app/config/config.yaml + template: monolith/deployment.yaml + - notExists: + path: spec.template.spec.containers[0].command + template: monolith/deployment.yaml + + - it: monolith appends monolith.extraArgs after the chart's own proxy arguments + set: + monolith.enabled: true + monolith.extraArgs: + - --detailed_debug + - --run_granian + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].args + value: + - proxy + - --port + - "4000" + - --config + - /app/config/config.yaml + - --detailed_debug + - --run_granian + + - it: monolith drops the --config pair when the chart does not create the ConfigMap + set: + monolith.enabled: true + gateway.config.create: false + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].args + value: + - proxy + - --port + - "4000" + - notExists: + path: spec.template.spec.volumes + + - it: monolith mounts the proxy config ConfigMap the chart renders under the monolith name + set: + monolith.enabled: true + gateway.config.proxy_config: + model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + templates: + - monolith/deployment.yaml + - gateway/configmap.yaml + asserts: + - equal: + path: metadata.name + value: RELEASE-NAME-litellm-config + template: gateway/configmap.yaml + - contains: + path: spec.template.spec.volumes + content: + name: gateway-config + configMap: + name: RELEASE-NAME-litellm-config + template: monolith/deployment.yaml + - contains: + path: spec.template.spec.containers[0].env + content: + name: CONFIG_FILE_PATH + value: /app/config/config.yaml + template: monolith/deployment.yaml + + - it: monolith mode renders exactly one Service and no component Services + set: + monolith.enabled: true + templates: + - monolith/service.yaml + - gateway/service.yaml + - backend/service.yaml + - ui/service.yaml + asserts: + - hasDocuments: + count: 1 + template: monolith/service.yaml + - hasDocuments: + count: 0 + template: gateway/service.yaml + - hasDocuments: + count: 0 + template: backend/service.yaml + - hasDocuments: + count: 0 + template: ui/service.yaml + - equal: + path: metadata.name + value: RELEASE-NAME-litellm + template: monolith/service.yaml + - equal: + path: spec.selector + value: + app.kubernetes.io/name: litellm + app.kubernetes.io/instance: RELEASE-NAME + app.kubernetes.io/component: proxy + template: monolith/service.yaml + - equal: + path: spec.ports[0].port + value: 4000 + template: monolith/service.yaml + + - it: monolith Service selector matches the monolith pod labels + set: + monolith.enabled: true + templates: + - monolith/deployment.yaml + asserts: + - equal: + path: spec.selector.matchLabels + value: + app.kubernetes.io/name: litellm + app.kubernetes.io/instance: RELEASE-NAME + app.kubernetes.io/component: proxy + - equal: + path: spec.template.metadata.labels["app.kubernetes.io/component"] + value: proxy + + - it: monolith reuses gateway resources, replicas, numWorkers and probes + set: + monolith.enabled: true + gateway.replicaCount: 3 + gateway.hpa.enabled: false + gateway.numWorkers: 6 + gateway.resources: + requests: + cpu: "2" + memory: 4Gi + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.replicas + value: 3 + - equal: + path: spec.template.spec.containers[0].resources.requests.cpu + value: "2" + - contains: + path: spec.template.spec.containers[0].env + content: + name: NUM_WORKERS + value: "6" + - equal: + path: spec.template.spec.containers[0].readinessProbe.httpGet.path + value: /health/readiness + - equal: + path: spec.template.spec.containers[0].livenessProbe.httpGet.path + value: /health/liveliness + + - it: monolith takes the rollout settings a litellm-helm install carried from gateway.minReadySeconds and gateway.strategy + set: + monolith.enabled: true + gateway.minReadySeconds: 60 + gateway.strategy: + type: RollingUpdate + rollingUpdate: + maxUnavailable: 0 + maxSurge: 1 + backend.minReadySeconds: 5 + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.minReadySeconds + value: 60 + - equal: + path: spec.strategy + value: + type: RollingUpdate + rollingUpdate: + maxUnavailable: 0 + maxSurge: 1 + + - it: monolith ignores backend and ui replica overrides + set: + monolith.enabled: true + gateway.hpa.enabled: false + gateway.replicaCount: 2 + backend.replicaCount: 9 + ui.replicaCount: 9 + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.replicas + value: 2 + + - it: monolith HPA targets the monolith Deployment and the gateway HPA is not rendered + set: + monolith.enabled: true + gateway.hpa.enabled: true + templates: + - monolith/hpa.yaml + - gateway/hpa.yaml + - backend/hpa.yaml + - ui/hpa.yaml + asserts: + - hasDocuments: + count: 1 + template: monolith/hpa.yaml + - hasDocuments: + count: 0 + template: gateway/hpa.yaml + - hasDocuments: + count: 0 + template: backend/hpa.yaml + - hasDocuments: + count: 0 + template: ui/hpa.yaml + - equal: + path: spec.scaleTargetRef + value: + apiVersion: apps/v1 + kind: Deployment + name: RELEASE-NAME-litellm + template: monolith/hpa.yaml + + - it: monolith KEDA ScaledObject targets the monolith Deployment + set: + monolith.enabled: true + gateway.hpa.enabled: false + gateway.keda.enabled: true + gateway.keda.prometheus.serverAddress: http://prometheus:9090 + gateway.keda.prometheus.targetRequestsPerSecond: 50 + templates: + - monolith/keda.yaml + - gateway/keda.yaml + asserts: + - hasDocuments: + count: 1 + template: monolith/keda.yaml + - hasDocuments: + count: 0 + template: gateway/keda.yaml + - equal: + path: spec.scaleTargetRef.name + value: RELEASE-NAME-litellm + template: monolith/keda.yaml + - equal: + path: spec.triggers[0].metadata.query + value: sum(rate(litellm_proxy_total_requests_metric_total{namespace="NAMESPACE",job="RELEASE-NAME-litellm-metrics"}[1m])) + template: monolith/keda.yaml + + - it: monolith PDB selects the monolith pods + set: + monolith.enabled: true + gateway.pdb.enabled: true + gateway.pdb.maxUnavailable: 1 + templates: + - monolith/poddisruptionbudget.yaml + - gateway/poddisruptionbudget.yaml + - backend/poddisruptionbudget.yaml + - ui/poddisruptionbudget.yaml + asserts: + - hasDocuments: + count: 1 + template: monolith/poddisruptionbudget.yaml + - hasDocuments: + count: 0 + template: gateway/poddisruptionbudget.yaml + - hasDocuments: + count: 0 + template: backend/poddisruptionbudget.yaml + - hasDocuments: + count: 0 + template: ui/poddisruptionbudget.yaml + - equal: + path: spec.maxUnavailable + value: 1 + template: monolith/poddisruptionbudget.yaml + - equal: + path: spec.selector.matchLabels["app.kubernetes.io/component"] + value: proxy + template: monolith/poddisruptionbudget.yaml + + - it: monolith metrics sidecar, metrics Service and ServiceMonitor follow the monolith name + set: + monolith.enabled: true + gateway.metricsServer.enabled: true + gateway.serviceMonitor.enabled: true + templates: + - monolith/deployment.yaml + - monolith/service-metrics.yaml + - monolith/servicemonitor.yaml + - gateway/service-metrics.yaml + - gateway/servicemonitor.yaml + asserts: + - hasDocuments: + count: 0 + template: gateway/service-metrics.yaml + - hasDocuments: + count: 0 + template: gateway/servicemonitor.yaml + - equal: + path: metadata.name + value: RELEASE-NAME-litellm-metrics + template: monolith/service-metrics.yaml + - equal: + path: spec.selector["app.kubernetes.io/component"] + value: proxy + template: monolith/service-metrics.yaml + - equal: + path: spec.selector.matchLabels["app.kubernetes.io/component"] + value: proxy + template: monolith/servicemonitor.yaml + - equal: + path: spec.template.spec.containers[1].name + value: metrics + template: monolith/deployment.yaml + - equal: + path: spec.template.spec.containers[1].args[0] + value: metrics + template: monolith/deployment.yaml + + - it: monolith collector sidecar runs the collector dispatcher from the same image + set: + monolith.enabled: true + gateway.collector.enabled: true + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[1].name + value: collector + - equal: + path: spec.template.spec.containers[1].args + value: + - collector + - equal: + path: spec.template.spec.containers[1].image + value: ghcr.io/berriai/litellm:1.104.0 + - contains: + path: spec.template.spec.containers[0].env + content: + name: LITELLM_COLLECTOR_ENABLED + value: "true" + + - it: monolith renders only the gateway ServiceAccount and runs the pod with it + set: + monolith.enabled: true + serviceAccounts.gateway.create: true + serviceAccounts.backend.create: true + serviceAccounts.ui.create: true + templates: + - serviceaccount.yaml + - monolith/deployment.yaml + asserts: + - hasDocuments: + count: 1 + template: serviceaccount.yaml + - equal: + path: metadata.name + value: RELEASE-NAME-litellm-gateway + template: serviceaccount.yaml + - equal: + path: spec.template.spec.serviceAccountName + value: RELEASE-NAME-litellm-gateway + template: monolith/deployment.yaml + + - it: the migrations Job renders the same way in monolith mode + set: + monolith.enabled: true + template: migrations-job.yaml + asserts: + - isKind: + of: Job + - equal: + path: spec.template.spec.containers[0].args + value: + - migrations + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + + - it: monolith gets the writer user under both names the proxy CLI and the componentized entrypoints read + set: + monolith.enabled: true + template: monolith/deployment.yaml + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_USER + valueFrom: + secretKeyRef: + name: litellm-writer-secret + key: username + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_USERNAME + valueFrom: + secretKeyRef: + name: litellm-writer-secret + key: username + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_NAME + value: litellm + + - it: disabling every component without monolith renders no application workload + set: + gateway.enabled: false + backend.enabled: false + ui.enabled: false + templates: + - monolith/deployment.yaml + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + asserts: + - hasDocuments: + count: 0 + + - it: monolith renders even when gateway.enabled is false + set: + monolith.enabled: true + gateway.enabled: false + templates: + - monolith/deployment.yaml + - monolith/service.yaml + asserts: + - hasDocuments: + count: 1 diff --git a/helm/litellm/tests/no_bundled_datastores_tests.yaml b/helm/litellm/tests/no_bundled_datastores_tests.yaml new file mode 100644 index 00000000000..dbafa5a9a11 --- /dev/null +++ b/helm/litellm/tests/no_bundled_datastores_tests.yaml @@ -0,0 +1,54 @@ +suite: the chart ships no PostgreSQL or Redis of its own +templates: + - gateway/deployment.yaml + - gateway/configmap.yaml + - migrations-job.yaml +tests: + - it: rejects the retired postgresql block instead of silently ignoring it + set: + postgresql.enabled: true + postgresql.auth.password: super-secret + asserts: + - failedTemplate: {} + + - it: rejects the retired redis.enabled flag instead of silently ignoring it + set: + redis.enabled: true + asserts: + - failedTemplate: {} + + - it: wires the writer straight from database.writer with no subchart in between + template: gateway/deployment.yaml + set: + database.writer.host: postgres.example.com + database.writer.dbname: proxydb + database.writer.passwordSecret.name: pg-secret + asserts: + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_HOST + value: postgres.example.com + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_NAME + value: proxydb + - contains: + path: spec.template.spec.containers[0].env + content: + name: DATABASE_PASSWORD + valueFrom: + secretKeyRef: + name: pg-secret + key: password + + - it: keeps the migrations Job a pre-install hook whenever the Helm hook is enabled + template: migrations-job.yaml + set: + database.writer.host: postgres.example.com + database.writer.dbname: litellm + asserts: + - equal: + path: metadata.annotations["helm.sh/hook"] + value: pre-install,pre-upgrade diff --git a/helm/litellm/tests/rollout_strategy_tests.yaml b/helm/litellm/tests/rollout_strategy_tests.yaml index b12e2073c7c..af6e41d038e 100644 --- a/helm/litellm/tests/rollout_strategy_tests.yaml +++ b/helm/litellm/tests/rollout_strategy_tests.yaml @@ -64,3 +64,39 @@ tests: - notExists: path: spec.strategy template: ui/deployment.yaml + + - it: leaves minReadySeconds to the Kubernetes default when unset + asserts: + - notExists: + path: spec.minReadySeconds + + - it: renders the configured minReadySeconds on each deployment + set: + gateway.minReadySeconds: 60 + backend.minReadySeconds: 30 + ui.minReadySeconds: 10 + asserts: + - equal: + path: spec.minReadySeconds + value: 60 + template: gateway/deployment.yaml + - equal: + path: spec.minReadySeconds + value: 30 + template: backend/deployment.yaml + - equal: + path: spec.minReadySeconds + value: 10 + template: ui/deployment.yaml + + - it: renders an explicit minReadySeconds of 0 instead of dropping it + set: + gateway.minReadySeconds: 0 + asserts: + - equal: + path: spec.minReadySeconds + value: 0 + template: gateway/deployment.yaml + - notExists: + path: spec.minReadySeconds + template: backend/deployment.yaml diff --git a/helm/litellm/tests/shared_image_tests.yaml b/helm/litellm/tests/shared_image_tests.yaml new file mode 100644 index 00000000000..1e7837b1973 --- /dev/null +++ b/helm/litellm/tests/shared_image_tests.yaml @@ -0,0 +1,264 @@ +suite: test the shared image and the entrypoint dispatcher args +templates: + - gateway/deployment.yaml + - gateway/configmap.yaml + - backend/deployment.yaml + - ui/deployment.yaml + - monolith/deployment.yaml + - migrations-job.yaml + - templates/tests/test-connection.yaml +values: + - ./values/required.yaml +tests: + - it: every componentized container runs the shared image at the chart appVersion + templates: + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + - migrations-job.yaml + - templates/tests/test-connection.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + template: backend/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + template: ui/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + template: migrations-job.yaml + - equal: + path: spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0 + template: templates/tests/test-connection.yaml + + - it: image.repository and image.tag are applied to every container, sidecars included + set: + image.repository: registry.example.com/litellm + image.tag: v9.9.9 + image.pullPolicy: Always + gateway.metricsServer.enabled: true + gateway.collector.enabled: true + templates: + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + - migrations-job.yaml + - templates/tests/test-connection.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].image + value: registry.example.com/litellm:v9.9.9 + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[1].image + value: registry.example.com/litellm:v9.9.9 + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[2].image + value: registry.example.com/litellm:v9.9.9 + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[0].imagePullPolicy + value: Always + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: registry.example.com/litellm:v9.9.9 + template: backend/deployment.yaml + - equal: + path: spec.template.spec.containers[0].imagePullPolicy + value: Always + template: backend/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: registry.example.com/litellm:v9.9.9 + template: ui/deployment.yaml + - equal: + path: spec.template.spec.containers[0].imagePullPolicy + value: Always + template: ui/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: registry.example.com/litellm:v9.9.9 + template: migrations-job.yaml + - equal: + path: spec.template.spec.containers[0].imagePullPolicy + value: Always + template: migrations-job.yaml + - equal: + path: spec.containers[0].image + value: registry.example.com/litellm:v9.9.9 + template: templates/tests/test-connection.yaml + + - it: the monolith and its sidecars run the shared image + set: + monolith.enabled: true + image.tag: v9.9.9 + gateway.metricsServer.enabled: true + gateway.collector.enabled: true + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9 + - equal: + path: spec.template.spec.containers[1].image + value: ghcr.io/berriai/litellm:v9.9.9 + - equal: + path: spec.template.spec.containers[2].image + value: ghcr.io/berriai/litellm:v9.9.9 + + - it: image.digest renders as repository:tag@digest everywhere + set: + image.tag: v9.9.9 + image.digest: sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + templates: + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + - migrations-job.yaml + - templates/tests/test-connection.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + template: backend/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + template: ui/deployment.yaml + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + template: migrations-job.yaml + - equal: + path: spec.containers[0].image + value: ghcr.io/berriai/litellm:v9.9.9@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + template: templates/tests/test-connection.yaml + + - it: image.digest with an empty tag falls back to the appVersion tag + set: + image.digest: sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + monolith.enabled: true + template: monolith/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].image + value: ghcr.io/berriai/litellm:1.104.0@sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef + + - it: componentized containers select their component through args and never set command + set: + gateway.metricsServer.enabled: true + gateway.collector.enabled: true + templates: + - gateway/deployment.yaml + - backend/deployment.yaml + - ui/deployment.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].args + value: + - gateway + - --host + - 0.0.0.0 + - --port + - "4000" + template: gateway/deployment.yaml + - notExists: + path: spec.template.spec.containers[0].command + template: gateway/deployment.yaml + - notExists: + path: spec.template.spec.containers[1].command + template: gateway/deployment.yaml + - notExists: + path: spec.template.spec.containers[2].command + template: gateway/deployment.yaml + - equal: + path: spec.template.spec.containers[0].args + value: + - backend + - --host + - 0.0.0.0 + - --port + - "4001" + template: backend/deployment.yaml + - notExists: + path: spec.template.spec.containers[0].command + template: backend/deployment.yaml + - equal: + path: spec.template.spec.containers[0].args + value: + - ui + template: ui/deployment.yaml + - notExists: + path: spec.template.spec.containers[0].command + template: ui/deployment.yaml + + - it: the migrations Job runs the migrations dispatcher and nothing else + template: migrations-job.yaml + asserts: + - equal: + path: spec.template.spec.containers[0].args + value: + - migrations + - notExists: + path: spec.template.spec.containers[0].command + + - it: the Helm test pod polls the workload Service readiness endpoint through the image entrypoint + template: templates/tests/test-connection.yaml + asserts: + - isKind: + of: Pod + - equal: + path: metadata.annotations["helm.sh/hook"] + value: test + - equal: + path: spec.containers[0].env[0] + value: + name: LITELLM_READINESS_URL + value: http://RELEASE-NAME-litellm-gateway:4000/health/readiness + - equal: + path: spec.containers[0].args[0] + value: python + - equal: + path: spec.containers[0].args[1] + value: -c + - matchRegex: + path: spec.containers[0].args[2] + pattern: LITELLM_READINESS_URL + - matchRegex: + path: spec.containers[0].args[2] + pattern: 'db != "connected"' + - matchRegex: + path: spec.containers[0].args[2] + pattern: sys\.exit\(1\) + - notExists: + path: spec.containers[0].command + + - it: the Helm test pod targets the monolith Service in monolith mode + set: + monolith.enabled: true + template: templates/tests/test-connection.yaml + asserts: + - equal: + path: spec.containers[0].env[0].value + value: http://RELEASE-NAME-litellm:4000/health/readiness + + - it: the removed per-component image blocks are rejected by the values schema + set: + gateway.image.tag: v1 + template: gateway/deployment.yaml + asserts: + - failedTemplate: {} diff --git a/helm/litellm/values.schema.json b/helm/litellm/values.schema.json new file mode 100644 index 00000000000..3aafbe68fe6 --- /dev/null +++ b/helm/litellm/values.schema.json @@ -0,0 +1,183 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "title": "LiteLLM chart values", + "type": "object", + "additionalProperties": true, + "properties": { + "image": { + "type": "object", + "additionalProperties": false, + "required": [ + "repository" + ], + "properties": { + "repository": { + "type": "string", + "minLength": 1 + }, + "tag": { + "type": "string" + }, + "pullPolicy": { + "type": "string", + "enum": [ + "Always", + "IfNotPresent", + "Never" + ] + }, + "digest": { + "type": "string", + "pattern": "^(sha256:[a-f0-9]{64})?$" + } + } + }, + "imagePullSecrets": { + "type": "array", + "items": { + "type": "object", + "required": [ + "name" + ], + "properties": { + "name": { + "type": "string" + } + } + } + }, + "monolith": { + "type": "object", + "additionalProperties": false, + "properties": { + "enabled": { + "type": "boolean" + }, + "extraArgs": { + "type": "array", + "items": { + "type": "string" + } + } + } + }, + "gateway": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": { + "type": "boolean" + }, + "replicaCount": { + "type": [ + "integer", + "null" + ], + "minimum": 0 + }, + "numWorkers": { + "type": [ + "integer", + "null" + ], + "minimum": 1 + }, + "image": false + } + }, + "backend": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": { + "type": "boolean" + }, + "replicaCount": { + "type": [ + "integer", + "null" + ], + "minimum": 0 + }, + "image": false + } + }, + "ui": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": { + "type": "boolean" + }, + "replicaCount": { + "type": [ + "integer", + "null" + ], + "minimum": 0 + }, + "image": false + } + }, + "migrationJob": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": { + "type": "boolean" + }, + "image": false + } + }, + "masterKey": { + "type": "object", + "additionalProperties": true, + "properties": { + "secretName": { + "type": "string" + }, + "secretKey": { + "type": "string" + }, + "generate": { + "type": "boolean" + } + } + }, + "postgresql": false, + "redis": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": false, + "cluster": { + "type": "boolean" + }, + "host": { + "type": "string" + }, + "port": { + "type": [ + "integer", + "string" + ] + } + } + }, + "ingress": { + "type": "object", + "additionalProperties": true, + "properties": { + "enabled": { + "type": "boolean" + } + } + }, + "extraResources": { + "type": "array", + "items": { + "type": "object" + } + } + } +} diff --git a/helm/litellm/values.yaml b/helm/litellm/values.yaml index 2c0c7151a32..5eaa2845420 100644 --- a/helm/litellm/values.yaml +++ b/helm/litellm/values.yaml @@ -1,10 +1,64 @@ # LiteLLM helm chart values +# +# Two deployment modes, both from the single image below: +# +# componentized (default): `gateway.enabled`, `backend.enabled` and +# `ui.enabled` each render their own Deployment, Service, HPA and PDB, so +# the LLM data plane, the management API and the static dashboard scale +# independently. Their containers run `args: [gateway|backend|ui, ...]`. +# +# monolith: `monolith.enabled: true` renders ONE Deployment and Service +# running the full proxy (`args: [proxy, ...]`): gateway routes, +# management routes and the Admin UI served by the proxy itself. The +# gateway, backend and ui Deployments and Services are not rendered and +# the Ingress sends every path to the monolith Service. The monolith is +# configured through the `gateway.*` values (config, resources, probes, +# securityContext, hpa, pdb, keda, collector, metricsServer, volumes, +# env), so a monolith install only flips this flag. See `monolith` below +# for the precedence rules. nameOverride: "" fullnameOverride: "" +# The one image every container in this chart runs: gateway, backend, ui, +# monolith, the migrations Job, the metrics and collector sidecars and the +# Helm test pod. Its entrypoint dispatches on the first container argument +# (proxy, gateway, backend, ui, migrations, metrics, collector), which is why +# the templates only ever set `args` and never `command`: overriding the +# entrypoint would drop the wrappers it applies (ddtrace, Prometheus +# multiprocess dir cleanup). `tag` defaults to the chart's appVersion. When +# `digest` is set the image renders as `repository:tag@digest`, so the tag is +# informational and the digest pins the bytes. Also mirrored at +# docker.litellm.ai/berriai/litellm. +image: + repository: ghcr.io/berriai/litellm + tag: "" + pullPolicy: IfNotPresent + digest: "" + imagePullSecrets: [] +# Monolith mode. When enabled the chart renders a single Deployment named +# `-litellm` running `args: [proxy, --port, 4000, --config, +# /app/config/config.yaml]` (the --config pair only when gateway.config.create +# is true) plus `extraArgs`, and a Service of the same name. Precedence when +# this is on: +# - gateway.enabled / backend.enabled / ui.enabled are ignored: no +# component Deployment, Service, HPA, PDB or ServiceMonitor is rendered +# - every `backend.*` and `ui.*` value is ignored, including their +# ServiceAccounts; the monolith runs as serviceAccounts.gateway +# - every `gateway.*` value applies to the monolith pod instead of a gateway +# pod: config, numWorkers, resources, probes, securityContext, +# hpa / keda / pdb, metricsServer, collector, volumes, env, scheduling +# - ingress.extraPaths entries keep their `service` field but every path, +# built in or extra, targets the monolith Service +# - the migrations Job renders exactly as in componentized mode +monolith: + enabled: false + # Extra CLI arguments appended after the chart's own proxy arguments, e.g. + # ["--detailed_debug"] or ["--run_granian"]. + extraArgs: [] + # Optional Ingress wiring the three component Services behind a single L7 # entrypoint. Required when serving the static UI bundle over the network. ingress: @@ -82,10 +136,8 @@ serviceAccounts: # LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...). Disable if your # pipeline runs migrations out-of-band. # -# Uses a dedicated `litellm-migrations` image (prisma CLI + the migration -# files from `litellm-proxy-extras`) instead of the backend image, so the -# Job doesn't drag in the rest of the proxy and doesn't run `prisma -# generate` — the migration engine doesn't need the generated client. +# Runs the shared image with `args: [migrations]`, which the entrypoint +# dispatches to the prisma migration runner without importing the proxy. migrationJob: enabled: true # Which controller is responsible for running the Job. @@ -159,21 +211,26 @@ migrationJob: nodeSelector: {} tolerations: [] affinity: {} - image: - repository: ghcr.io/berriai/litellm-migrations - tag: "" # defaults to .Chart.AppVersion - pullPolicy: IfNotPresent + # Extra init containers and sidecars on the Job pod, rendered through `tpl`. + extraInitContainers: [] + extraContainers: [] # Extra env appended to the migration container. The migration entrypoint # uses the v2 resolver by default (no diff-and-force recovery — avoids the # schema thrashing seen during rolling deploys). To opt back into the v1 # resolver, append `- name: USE_V2_MIGRATION_RESOLVER` / `value: "false"`. extraEnv: [] -# Required: a master key used by gateway + backend to mint/verify proxy tokens. -# Must reference an existing Secret. +# Master key used by the proxy to mint and verify virtual keys. Reference an +# existing Secret through `secretName` / `secretKey`, or set `generate: true` +# and leave `secretName` empty to have the chart create +# `-litellm-masterkey` with a random `sk-...` key under `secretKey`. +# The generated Secret is looked up on upgrade so the key never rotates by +# accident (`helm template` cannot look anything up and renders a fresh key +# each time; that is expected). The chart never accepts an inline key. masterKey: secretName: litellm-master-key-secret # name of a Secret containing the master key secretKey: master-key + generate: false # Optional: enterprise billable-request metering. When enabled, the gateway and # backend count successful requests to inference, MCP, and A2A endpoints and push @@ -194,7 +251,8 @@ billingMetrics: caSecretName: "" # existing Secret holding ca.crt exportIntervalMs: "" # push cadence; the proxy defaults to 60000 -# External Postgres connection. +# Postgres connection. The chart ships no database; point `writer` at your +# own PostgreSQL (managed or self hosted) and its credentials Secret. database: writer: host: "" @@ -250,23 +308,22 @@ database: maxDbConnections: 20 maxClientConn: 1000 -# Optional Redis. Leave host empty to disable. +# Redis is the proxy's coordination store: cross-pod tpm/rpm rate limits, +# spend tracking, and the pod lock manager. The chart emits REDIS_HOST / +# REDIS_PORT / REDIS_PASSWORD, which the proxy picks up through its +# coordination Redis env fallback. Response caching is separate and off unless +# you enable it in `proxy_config.litellm_settings.cache`. # -# This is the proxy's coordination store: cross-pod tpm/rpm rate limits, spend -# tracking, and the pod lock manager. The chart emits REDIS_HOST / REDIS_PORT / -# REDIS_PASSWORD, which the proxy picks up through its coordination Redis env -# fallback. Response caching is separate and off unless you enable it in -# `proxy_config.litellm_settings.cache`. +# Set `host` (and `passwordSecret.name` when auth is required). Set +# `cluster: true` for Redis Cluster mode (AWS ElastiCache Cluster, self-hosted +# Redis Cluster): the chart emits REDIS_CLUSTER_NODES from `host` / `port` as +# the single seed and the client discovers the rest from CLUSTER SLOTS. The +# chart ships no Redis of its own. # # For full control, define `general_settings.coordination_redis` in # `proxy_config` (host/port/password/username/url/ssl/startup_nodes/ # sentinel_nodes/sentinel_password/service_name, each accepting os.environ/VAR # refs). An explicit block overrides these env vars. -# -# Set `cluster: true` for Redis Cluster mode (e.g. AWS ElastiCache Cluster, -# self-hosted Redis Cluster). The chart emits REDIS_CLUSTER_NODES from -# `host` / `port` as the single seed; the cluster client discovers the -# remaining nodes from CLUSTER SLOTS at startup. redis: cluster: false host: "" @@ -275,12 +332,17 @@ redis: name: "" # Leave empty for auth-less Redis passwordKey: password +# Arbitrary extra Kubernetes manifests rendered verbatim with the release, +# e.g. a NetworkPolicy or an ExternalSecret the proxy needs. +extraResources: [] + # ---------- gateway (LLM data plane) ---------- +# In monolith mode this whole block configures the monolith pod instead. gateway: enabled: true logLevel: INFO - # Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS, - # consumed by the gateway image entrypoint. Default is 1. + # Number of worker processes per pod. Sets NUM_WORKERS, read by the gateway + # launcher and by the proxy's --num_workers default. Default is 1. numWorkers: 1 extraEnv: [] # Add extra environment variables to the gateway envConfigMaps: [] # Add extra environment variables to the gateway from config maps @@ -299,7 +361,6 @@ gateway: # scrape never runs on an inference worker. Adds a `metrics` port to the pod # and a dedicated ClusterIP `-metrics` Service; point your scrape # config at it. The port has no virtual-key auth: keep it off public ingress. - # Needs the gateway image v1.101.0 or newer. metricsServer: enabled: false port: 4001 @@ -355,13 +416,13 @@ gateway: # ContainerResource metric of the `gateway` container only, so the # sidecar's CPU never drives inference replicas. Needs Kubernetes 1.30+. scaleOnGatewayContainerCpu: true - image: - repository: ghcr.io/berriai/litellm-gateway - tag: "" # defaults to .Chart.AppVersion - pullPolicy: IfNotPresent service: type: ClusterIP port: 4000 + annotations: {} + # For LoadBalancer Services on clusters with several load balancer + # implementations. + loadBalancerClass: "" resources: requests: cpu: "1" @@ -435,6 +496,36 @@ gateway: # counted when a response completes, so TPS trails long streams. targetRequestsPerSecond: "" targetTokensPerSecond: "" + # KEDA ScaledObject as an alternative to the HPA above. Mutually exclusive + # with hpa.enabled (two autoscalers on one Deployment fight each other), so + # set `hpa.enabled: false` when turning this on. Needs KEDA installed in the + # cluster. `triggers` are rendered verbatim; the two optional Prometheus + # workload targets below add prometheus triggers on the same counters the + # HPA workload metrics use, scraped through the metrics Service (enable + # metricsServer and serviceMonitor). + keda: + enabled: false + minReplicaCount: 1 + maxReplicaCount: 10 + pollingInterval: 30 + cooldownPeriod: 300 + # Optional; rendered verbatim under spec.advanced + advanced: {} + # Optional; rendered verbatim under spec.fallback + fallback: {} + # Custom triggers, rendered verbatim, e.g. + # - type: cpu + # metricType: Utilization + # metadata: + # value: "70" + triggers: [] + prometheus: + # Prometheus server URL, required when either workload target is set. + serverAddress: "" + # Per pod averages, same units as hpa.targetRequestsPerSecond / + # hpa.targetTokensPerSecond. Empty leaves the trigger out. + targetRequestsPerSecond: "" + targetTokensPerSecond: "" # PodDisruptionBudget for the gateway pods. Set exactly one of # `minAvailable` / `maxUnavailable` (minAvailable wins if both are set; # enabling without either falls back to `maxUnavailable: 1`). Disabled by @@ -444,6 +535,13 @@ gateway: enabled: false minAvailable: "" maxUnavailable: "" + # Extra annotations and labels on the Deployment object itself (not the + # pods), e.g. for a GitOps controller or a reloader. + deploymentAnnotations: {} + deploymentLabels: {} + # Seconds a new pod must be ready before the rollout treats it as + # available. Empty inherits the Kubernetes default of 0. + minReadySeconds: "" podAnnotations: {} # Extra pod labels, merged into the chart's selector labels. Do not # re-declare `app.kubernetes.io/name` / `instance` / `component` here: they @@ -470,7 +568,16 @@ gateway: # egress proxy. Rendered through `tpl`, so entries may reference chart # values and release metadata. extraContainers: [] + # Init containers, rendered through `tpl` the same way. + extraInitContainers: [] # Container lifecycle hooks (postStart / preStop) for the gateway container. + # + # Prefer the proxy's /health/drain preStop hook over a fixed `sleep`: it + # marks the pod NotReady and blocks only until in-flight requests finish + # (bounded by GRACEFUL_SHUTDOWN_TIMEOUT). It is off by default; enable it + # with general_settings.enable_drain_endpoint: true and, when the health + # port is reachable from other pods, set general_settings.drain_endpoint_token + # and send the same value on the X-Drain-Token header from the hook. lifecycle: {} # Grace period the kubelet allows between SIGTERM and SIGKILL. Leave empty # to inherit the Kubernetes default of 30s. Set it a few seconds above the @@ -500,13 +607,11 @@ backend: volumes: [] # Additional volumeMounts on the backend container. volumeMounts: [] - image: - repository: ghcr.io/berriai/litellm-backend - tag: "" - pullPolicy: IfNotPresent service: type: ClusterIP port: 4001 + annotations: {} + loadBalancerClass: "" resources: requests: cpu: "1" @@ -543,12 +648,16 @@ backend: enabled: false minAvailable: "" maxUnavailable: "" + deploymentAnnotations: {} + deploymentLabels: {} + minReadySeconds: "" podAnnotations: {} # Same shape as the gateway blocks of the same name. podLabels: {} podSecurityContext: {} securityContext: {} extraContainers: [] + extraInitContainers: [] lifecycle: {} terminationGracePeriodSeconds: "" nodeSelector: {} @@ -568,13 +677,11 @@ ui: volumes: [] # Additional volumeMounts on the ui container. volumeMounts: [] - image: - repository: ghcr.io/berriai/litellm-ui - tag: "" - pullPolicy: IfNotPresent service: type: ClusterIP port: 3000 + annotations: {} + loadBalancerClass: "" # The dashboard expects to know where to reach the backend API. Set this to # the externally-routable URL (typically the ingress host + /api or similar). backendUrl: "" @@ -611,17 +718,19 @@ ui: enabled: false minAvailable: "" maxUnavailable: "" + deploymentAnnotations: {} + deploymentLabels: {} + minReadySeconds: "" podAnnotations: {} # Same shape as the gateway blocks of the same name. The nginx runtime # writes its pid, cache, and proxy temp files under /tmp, so it boots as # any (arbitrary, non-root) uid; `securityContext.readOnlyRootFilesystem: - # true` here needs an emptyDir volume mounted over /tmp. Images before - # the /tmp move instead need emptyDirs over /var/cache/nginx and /run to - # run as a non-root uid at all. + # true` here needs an emptyDir volume mounted over /tmp. podLabels: {} podSecurityContext: {} securityContext: {} extraContainers: [] + extraInitContainers: [] lifecycle: {} terminationGracePeriodSeconds: "" nodeSelector: {} diff --git a/scripts/tpm_headline_test.sh b/scripts/tpm_headline_test.sh index a2f4063d311..ef6750e2433 100755 --- a/scripts/tpm_headline_test.sh +++ b/scripts/tpm_headline_test.sh @@ -12,7 +12,7 @@ # Post-PR: only ~1–2 fit under tpm_limit=100, rest return 429. # # Setup (separate terminal): -# kubectl port-forward -n litellm svc/yassin-veks-litellm-helm 4000:4000 +# kubectl port-forward -n litellm svc/litellm-gateway 4000:4000 # # Run: # bash scripts/tpm_headline_test.sh