diff --git a/backend/Dockerfile b/backend/Dockerfile new file mode 100644 index 00000000000..c08014fc0ef --- /dev/null +++ b/backend/Dockerfile @@ -0,0 +1,83 @@ +ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:31da6565f35af6401031c1d7aa91dc84ac76c5c48edd17fb90f0ed9e3173c7a9 +ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:31da6565f35af6401031c1d7aa91dc84ac76c5c48edd17fb90f0ed9e3173c7a9 +ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a + +FROM $UV_IMAGE AS uvbin + +# ---------- Builder ---------- +FROM $LITELLM_BUILD_IMAGE AS builder + +WORKDIR /app +USER root + +COPY --from=uvbin /uv /uvx /usr/local/bin/ + +RUN apk add --no-cache bash gcc python3 python3-dev openssl openssl-dev libsndfile + +# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start. +# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a +# BuildKit cache mount (different filesystem). +# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of +# silently pulling a managed interpreter. +ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ + UV_LINK_MODE=copy \ + UV_COMPILE_BYTECODE=1 \ + UV_PYTHON_DOWNLOADS=0 \ + PATH="/app/.venv/bin:${PATH}" + +# Stage 1 — install dependencies only. +RUN --mount=type=cache,target=/root/.cache/uv \ + --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ + --mount=type=bind,source=uv.lock,target=uv.lock \ + --mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \ + --mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \ + uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ + --extra proxy \ + --extra proxy-runtime \ + --extra extra_proxy \ + --extra semantic-router \ + --python python3 + +# Stage 2 — copy source and install the project + workspace members. +COPY . . + +RUN --mount=type=cache,target=/root/.cache/uv \ + uv sync --frozen --no-default-groups --no-editable \ + --extra proxy \ + --extra proxy-runtime \ + --extra extra_proxy \ + --extra semantic-router \ + --python python3 + +RUN mkdir -p /home/nonroot && \ + HOME=/home/nonroot prisma generate --schema=./schema.prisma && \ + chown -R nonroot:nonroot /home/nonroot/.cache + +# ---------- Runtime ---------- +FROM $LITELLM_RUNTIME_IMAGE AS runtime + +USER root + +RUN apk add --no-cache bash openssl tzdata python3 libsndfile libatomic + +# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with +# /home/nonroot. We run the backend as that user +WORKDIR /app +ENV HOME=/home/nonroot \ + PATH="/app/.venv/bin:${PATH}" \ + PYTHONPATH="/app" \ + PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 + +COPY --from=builder --chown=nonroot:nonroot /app /app +COPY --from=builder --chown=nonroot:nonroot /home/nonroot/.cache /home/nonroot/.cache + +RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ + find /app/.venv -type d -path "*/tornado/test" -delete + +USER nonroot + +EXPOSE 4001/tcp + +ENTRYPOINT ["uvicorn", "backend.main:app"] +CMD ["--host", "0.0.0.0", "--port", "4001"] diff --git a/backend/main.py b/backend/main.py new file mode 100644 index 00000000000..194085bed2d --- /dev/null +++ b/backend/main.py @@ -0,0 +1,37 @@ +"""UI backend entrypoint. + +Reuses the existing FastAPI app from `litellm.proxy.proxy_server` and trims its +route table to just the management/admin surface used by the dashboard. Purely +additive — no existing module is modified. + +Run with: + uvicorn backend.main:app --host 0.0.0.0 --port 4001 +""" + +from fastapi.routing import Mount + +# See gateway/main.py for why we mint IAM-signed DATABASE_URL(s) here before +# importing proxy_server. +from litellm.proxy.auth.rds_iam_token import init_iam_db_url_from_env + +init_iam_db_url_from_env() + +from litellm.proxy.proxy_server import app + +from backend.routes.allowlist import BACKEND_EXACT_PATHS, BACKEND_PATH_PREFIXES + + +def _is_backend_route(route) -> bool: + """Keep the route on the backend if its path is in the management surface.""" + path = getattr(route, "path", None) + if path is None: + return False + if isinstance(route, Mount): + # Static UI mounts are served by the dedicated UI container, not here. + return False + if path in BACKEND_EXACT_PATHS: + return True + return any(path.startswith(prefix) for prefix in BACKEND_PATH_PREFIXES) + + +app.router.routes = [r for r in app.router.routes if _is_backend_route(r)] diff --git a/backend/routes/__init__.py b/backend/routes/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/backend/routes/allowlist.py b/backend/routes/allowlist.py new file mode 100644 index 00000000000..d90afae30cd --- /dev/null +++ b/backend/routes/allowlist.py @@ -0,0 +1,133 @@ +"""Path allowlist for the UI backend (control plane) component. + +The backend exposes management/admin endpoints consumed by the UI: keys, users, +teams, orgs, customers, budgets, tags, workflows, model management, spend & +analytics, settings (router/cache/cost-tracking/fallbacks), SSO/onboarding, +audit logs, debug, enterprise admin, and UI bootstrap helpers (logo, favicon, +.well-known config). + +Anything LLM data-plane is dropped — those run on the gateway component. +""" + +BACKEND_PATH_PREFIXES: tuple[str, ...] = ( + # Identity / access + "/key/", + "/v2/key/", + "/user/", + "/v2/user/", + "/team/", + "/v2/team/", + "/organization/", + "/customer/", + "/end_user/", + "/sso/", + "/login", + "/v2/login", + "/v3/login", + "/logout", + "/token", + "/onboarding/", + "/audit", + "/oauth/", + "/invitation/", + "/jwt/", + # Models & routing config + "/model/", + "/v1/model/info", + "/v2/model/", + "/model_group", + "/model_access_group/", + "/model_hub/", + "/v1/access_group", + "/access_group/", + "/router/", + "/router_settings", + "/adaptive_router/", + "/fallback", + "/fallbacks", + "/cache_settings", + "/cost_tracking", + "/cost/", + "/credentials", + "/credential", + "/provider/budgets", + # Tools / agents (registry & policy admin) + "/v1/tool/", + "/v1/agents", + # Guardrails admin + "/v2/guardrails/", + # MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints + "/v1/mcp/", + "/test/", + "/{mcp_server_name}/", + # Budgets / tags / workflows / memory mgmt + "/budget/", + "/tag/", + "/workflow/", + "/v1/workflows/", + "/project/", + "/memory/", + "/mcp/", + # Spend / analytics + "/spend/", + "/analytics/", + "/global/", + "/user_agent", + "/usage/", + "/daily/", + # Caching admin + "/cache/", + "/caching/", + # Callbacks / hooks + "/active/callbacks", + "/callbacks", + "/team_callback", + # Alerting / email / IP allowlist + "/alerting/", + "/email/", + "/add/allowed_ip", + "/delete/allowed_ip", + "/get/", + # Enterprise admin + "/enterprise/", + # Debug / config / profiling + "/debug/", + "/config/", + "/memory-usage-in-mem-cache", + "/otel-spans", + "/lazy/", + "/in_product_nudges", + # Admin reload / schedule + "/reload/", + "/schedule/", + "/settings", + "/update/", + "/upload/", + # Dev / admin utilities + "/utils/", + # UI bootstrap helpers (assets the dashboard fetches) + "/get_logo_url", + "/get_image", + "/get_favicon", + "/.well-known/", + "/litellm/.well-known/", + "/ui_discovery/", + "/ui-config", + "/sso_settings", + "/public/", + "/robots.txt", + # Health (k8s probes) + "/health", +) + +BACKEND_EXACT_PATHS: frozenset[str] = frozenset( + { + "/", + "/routes", + "/openapi.json", + "/docs", + "/docs/oauth2-redirect", + "/redoc", + "/fallback/login", + } +) diff --git a/gateway/Dockerfile b/gateway/Dockerfile new file mode 100644 index 00000000000..a2ca3d3f83f --- /dev/null +++ b/gateway/Dockerfile @@ -0,0 +1,83 @@ +ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:31da6565f35af6401031c1d7aa91dc84ac76c5c48edd17fb90f0ed9e3173c7a9 +ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:31da6565f35af6401031c1d7aa91dc84ac76c5c48edd17fb90f0ed9e3173c7a9 +ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a + +FROM $UV_IMAGE AS uvbin + +# ---------- Builder ---------- +FROM $LITELLM_BUILD_IMAGE AS builder + +WORKDIR /app +USER root + +COPY --from=uvbin /uv /uvx /usr/local/bin/ + +RUN apk add --no-cache bash gcc python3 python3-dev openssl openssl-dev libsndfile + +# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start. +# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a +# BuildKit cache mount (different filesystem). +# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of +# silently pulling a managed interpreter. +ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ + UV_LINK_MODE=copy \ + UV_COMPILE_BYTECODE=1 \ + UV_PYTHON_DOWNLOADS=0 \ + PATH="/app/.venv/bin:${PATH}" + +# Stage 1 — install dependencies only. +RUN --mount=type=cache,target=/root/.cache/uv \ + --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ + --mount=type=bind,source=uv.lock,target=uv.lock \ + --mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \ + --mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \ + uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ + --extra proxy \ + --extra proxy-runtime \ + --extra extra_proxy \ + --extra semantic-router \ + --python python3 + +# Stage 2 — copy source and install the project + workspace members. +COPY . . + +RUN --mount=type=cache,target=/root/.cache/uv \ + uv sync --frozen --no-default-groups --no-editable \ + --extra proxy \ + --extra proxy-runtime \ + --extra extra_proxy \ + --extra semantic-router \ + --python python3 + +RUN mkdir -p /home/nonroot && \ + HOME=/home/nonroot prisma generate --schema=./schema.prisma && \ + chown -R nonroot:nonroot /home/nonroot/.cache + +# ---------- Runtime ---------- +FROM $LITELLM_RUNTIME_IMAGE AS runtime + +USER root + +RUN apk add --no-cache bash openssl tzdata python3 libsndfile libatomic + +# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with +# /home/nonroot. We run the proxy as that user. +WORKDIR /app +ENV HOME=/home/nonroot \ + PATH="/app/.venv/bin:${PATH}" \ + PYTHONPATH="/app" \ + PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 + +COPY --from=builder --chown=nonroot:nonroot /app /app +COPY --from=builder --chown=nonroot:nonroot /home/nonroot/.cache /home/nonroot/.cache + +RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ + find /app/.venv -type d -path "*/tornado/test" -delete + +USER nonroot + +EXPOSE 4000/tcp + +ENTRYPOINT ["sh", "-c", "exec uvicorn gateway.main:app --workers \"${NUM_WORKERS:-1}\" \"$@\"", "--"] +CMD ["--host", "0.0.0.0", "--port", "4000"] diff --git a/gateway/main.py b/gateway/main.py new file mode 100644 index 00000000000..8a37abf0c48 --- /dev/null +++ b/gateway/main.py @@ -0,0 +1,42 @@ +"""Gateway entrypoint. + +Reuses the existing FastAPI app from `litellm.proxy.proxy_server` and trims its +route table to just the LLM data-plane surface. The trim is purely additive — +no existing module is modified, the full app continues to work via the legacy +entrypoint (`litellm.proxy.proxy_server:app`). + +Run with: + uvicorn gateway.main:app --host 0.0.0.0 --port 4000 +""" + +from fastapi.routing import Mount + +# Mint RDS IAM tokens and assemble DATABASE_URL (+ DATABASE_URL_READ_REPLICA) +# before proxy_server imports spin up Prisma. The standard CLI flow does this +# in proxy_cli.py; we bypass proxy_cli by uvicorn'ing the app directly, so +# without this call Prisma initializes with the placeholder URL and every +# DB-needing endpoint returns "Database not connected". No-op when +# IAM_TOKEN_DB_AUTH is unset. +from litellm.proxy.auth.rds_iam_token import init_iam_db_url_from_env + +init_iam_db_url_from_env() + +from litellm.proxy.proxy_server import app + +from gateway.routes.allowlist import GATEWAY_EXACT_PATHS, GATEWAY_PATH_PREFIXES + + +def _is_gateway_route(route) -> bool: + """Keep the route on the gateway if its path is in the LLM data-plane surface.""" + path = getattr(route, "path", None) + if path is None: + return False + if isinstance(route, Mount): + # Gateway never serves the static UI or its asset bundles. + return False + if path in GATEWAY_EXACT_PATHS: + return True + return any(path.startswith(prefix) for prefix in GATEWAY_PATH_PREFIXES) + + +app.router.routes = [r for r in app.router.routes if _is_gateway_route(r)] diff --git a/gateway/routes/__init__.py b/gateway/routes/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/gateway/routes/allowlist.py b/gateway/routes/allowlist.py new file mode 100644 index 00000000000..cbbf55c9873 --- /dev/null +++ b/gateway/routes/allowlist.py @@ -0,0 +1,121 @@ +"""Path allowlist for the gateway component. + +The gateway exposes the LLM data-plane surface: chat/completions, embeddings, +audio, batches, files, fine-tuning, rerank, ocr, rag, video, search, image, +responses, vector stores, passthrough providers, realtime websockets, MCP +tool-call endpoints, and operational endpoints (/health, /metrics). + +Any path not listed here is dropped from the gateway process so management/UI +endpoints don't ride on the same pods. + +Versioned data-plane paths are enumerated explicitly rather than allowing a +blanket `/v1/` or `/v2/` prefix — those broad prefixes would otherwise also +match management routes like `/v1/access_group`, `/v1/tool/{tool_name}/logs`, +`/v2/key/info`, etc. +""" + +GATEWAY_PATH_PREFIXES: tuple[str, ...] = ( + # OpenAI-compatible data-plane surface (versioned + unversioned) + "/v1/chat/", + "/chat/", + "/v1/completions", + "/completions", + "/v1/embeddings", + "/embeddings", + "/v1/moderations", + "/moderations", + "/v1/audio/", + "/audio/", + "/v1/images/", + "/images/", + "/v1/files", + "/files", + "/v1/batches", + "/batches", + "/v1/fine_tuning/", + "/fine_tuning/", + "/v1/fine-tuning/", + "/fine-tuning/", + "/v1/responses", + "/responses", + "/v1/threads", + "/threads", + "/v1/assistants", + "/assistants", + "/v1/vector_stores", + "/vector_stores", + "/v1/indexes", + "/v1/models", + "/models", + "/openai/", + "/engines/", + # Anthropic / agentic data-plane surface + "/v1/messages", + "/messages", + "/v1/skills", + "/v1/a2a/", + # LiteLLM-native LLM surface + "/v1/rerank", + "/v2/rerank", + "/rerank", + "/v1/ocr", + "/ocr", + "/v1/rag/", + "/rag/", + "/v1/video", + "/v1/videos", + "/video/", + "/videos", + "/v1/search", + "/search", + "/v1/containers", + "/containers", + "/v1/evals", + "/v1/memory", + "/queue/chat/", + # Google data plane (v1beta is the Google AI Studio version) + "/v1beta/", + "/interactions", + # Provider passthrough + "/anthropic/", + "/azure/", + "/azure_ai/", + "/aws/", + "/bedrock/", + "/cohere/", + "/gemini/", + "/google/", + "/vertex_ai/", + "/vertex-ai/", + "/assemblyai/", + "/eu.assemblyai/", + "/langfuse/", + "/vllm/", + "/mistral/", + "/groq/", + "/voyage/", + "/cursor/", + "/milvus/", + "/openai_passthrough/", + # Dynamic provider / toolset passthrough (path templates) + "/{provider}/", + "/toolset/", + # Realtime / streaming + "/v1/realtime", + "/realtime", + # Health & ops + "/health", + "/metrics", +) + +GATEWAY_EXACT_PATHS: frozenset[str] = frozenset( + { + "/", + "/routes", + "/openapi.json", + "/docs", + "/docs/oauth2-redirect", + "/redoc", + "/test", + } +) diff --git a/helm/litellm/Chart.yaml b/helm/litellm/Chart.yaml new file mode 100644 index 00000000000..e67f5790c7e --- /dev/null +++ b/helm/litellm/Chart.yaml @@ -0,0 +1,8 @@ +apiVersion: v2 +name: litellm +description: LiteLLM componentized — gateway, UI backend, and UI as separate services +type: application +version: 0.1.0 +appVersion: "0.1.0" +annotations: + org.opencontainers.image.source: "https://github.com/BerriAI/litellm" diff --git a/helm/litellm/templates/NOTES.txt b/helm/litellm/templates/NOTES.txt new file mode 100644 index 00000000000..5b939fe480a --- /dev/null +++ b/helm/litellm/templates/NOTES.txt @@ -0,0 +1,49 @@ +LiteLLM componentized — release {{ .Release.Name }} in namespace {{ .Release.Namespace }}. + +Components: +{{- if .Values.gateway.enabled }} + - gateway : Service {{ include "litellm.gateway.fullname" . }} on port {{ .Values.gateway.service.port }} +{{- end }} +{{- if .Values.backend.enabled }} + - backend : Service {{ include "litellm.backend.fullname" . }} on port {{ .Values.backend.service.port }} +{{- end }} +{{- if .Values.ui.enabled }} + - ui : Service {{ include "litellm.ui.fullname" . }} on port {{ .Values.ui.service.port }} +{{- end }} + +Port-forward examples: + kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.gateway.fullname" . }} {{ .Values.gateway.service.port }} + kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.backend.fullname" . }} {{ .Values.backend.service.port }} + kubectl -n {{ .Release.Namespace }} port-forward svc/{{ include "litellm.ui.fullname" . }} {{ .Values.ui.service.port }} + +Reminders: + - Sensitive values come from Secret references only. Before installing, set: + - masterKey.secretName (Secret with the proxy master key) + - database.writer.{host,port,dbname} (writer connection pieces) + - database.writer.passwordSecret.{name,usernameKey,passwordKey} + (Secret holding the writer DB username + password) + - database.writer.useIAMAuth: true (optional — chart sets IAM_TOKEN_DB_AUTH=true and + omits DATABASE_PASSWORD / DATABASE_URL so the proxy + mints the URL from an IAM token at startup) + - database.reader.host (optional — enables read-replica routing; reader + .passwordSecret.name is required when set, unless + .useIAMAuth is true) + - database.reader.useIAMAuth: true (optional, requires database.writer.useIAMAuth: true — + chart emits DATABASE_*_READ_REPLICA env vars and + omits DATABASE_PASSWORD_READ_REPLICA / + DATABASE_URL_READ_REPLICA so the proxy mints the + reader URL from an IAM token at startup) + - redis.passwordSecret.name (optional — set when redis.host is provided and the + cache requires auth) + - redis.cluster: true (optional — chart sets REDIS_CLUSTER_NODES from + redis.host / redis.port so the proxy's Cache() + constructs a RedisClusterCache; the cluster client + discovers remaining nodes from CLUSTER SLOTS) + - Per-component extras (gateway / backend / ui): + - {component}.extraEnv / envConfigMaps / envSecrets (the latter two are lists of resource names → + envFrom configMapRef / secretRef) + - {component}.logLevel (renders as LITELLM_LOG) + - gateway.config.proxy_config (rendered into a ConfigMap and mounted at + /app/config/config.yaml; gateway reads it via + CONFIG_FILE_PATH) + - Enable ingress.enabled=true to dispatch / → ui, gateway data-plane prefixes → gateway, and the catch-all → backend. diff --git a/helm/litellm/templates/_helpers.tpl b/helm/litellm/templates/_helpers.tpl new file mode 100644 index 00000000000..b57487d5469 --- /dev/null +++ b/helm/litellm/templates/_helpers.tpl @@ -0,0 +1,232 @@ +{{/* +Common naming + label helpers shared by gateway, backend, and ui templates. +*/}} + +{{- define "litellm.name" -}} +{{- default .Chart.Name .Values.nameOverride | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "litellm.fullname" -}} +{{- if .Values.fullnameOverride -}} +{{- .Values.fullnameOverride | trunc 63 | trimSuffix "-" -}} +{{- else -}} +{{- $name := default .Chart.Name .Values.nameOverride -}} +{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" -}} +{{- end -}} +{{- end -}} + +{{- define "litellm.gateway.fullname" -}} +{{- printf "%s-gateway" (include "litellm.fullname" .) | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "litellm.backend.fullname" -}} +{{- printf "%s-backend" (include "litellm.fullname" .) | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "litellm.ui.fullname" -}} +{{- printf "%s-ui" (include "litellm.fullname" .) | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "litellm.commonLabels" -}} +app.kubernetes.io/name: {{ include "litellm.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +app.kubernetes.io/managed-by: {{ .Release.Service }} +helm.sh/chart: {{ printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" }} +{{- end -}} + +{{/* +Per-component selector labels — used in both Service selectors and Deployment matchLabels. +*/}} +{{- define "litellm.gateway.selectorLabels" -}} +app.kubernetes.io/name: {{ include "litellm.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +app.kubernetes.io/component: gateway +{{- end -}} + +{{- define "litellm.backend.selectorLabels" -}} +app.kubernetes.io/name: {{ include "litellm.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +app.kubernetes.io/component: backend +{{- end -}} + +{{- define "litellm.ui.selectorLabels" -}} +app.kubernetes.io/name: {{ include "litellm.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +app.kubernetes.io/component: ui +{{- end -}} + +{{/* +Shared ServiceAccount name used by all three component Deployments. When +`serviceAccount.create` is true and `serviceAccount.name` is empty, default +to the chart fullname. When `create` is false, fall back to the provided +name or the namespace's `default` SA. +*/}} +{{- define "litellm.serviceAccountName" -}} +{{- if .Values.serviceAccount.create -}} +{{ default (include "litellm.fullname" .) .Values.serviceAccount.name }} +{{- else -}} +{{ default "default" .Values.serviceAccount.name }} +{{- end -}} +{{- end -}} + +{{/* +Master-key + database + redis env block — shared by gateway, backend, and the +migrations Job. + +Invoke with a dict: `(dict "root" $ "component" .Values.gateway)`. `root` is +the chart context (needed for .Values), `component` selects which component's +`extraEnv` / `logLevel` to render. + +Sensitive values (master key, DB username + password, Redis password) come +only from referenced Secrets; the chart never accepts inline values for them. + +DATABASE_URL is assembled at pod startup via Kubernetes `$(VAR)` env +substitution, so the unencoded password is never written to the Pod spec. +Kubernetes only substitutes vars declared earlier in the same container's +env list, so DATABASE_USER / DATABASE_PASSWORD must precede DATABASE_URL. + +When `database.writer.useIAMAuth: true`, the chart injects +IAM_TOKEN_DB_AUTH=true and omits DATABASE_PASSWORD / DATABASE_URL — the +proxy's entrypoint (litellm/proxy/auth/rds_iam_token.py:190) then mints +the URL from DATABASE_HOST/PORT/USER/NAME plus an IAM token. + +When `database.reader.useIAMAuth: true`, the chart emits +DATABASE_HOST_READ_REPLICA / DATABASE_PORT_READ_REPLICA / +DATABASE_NAME_READ_REPLICA (plus DATABASE_USER_READ_REPLICA when a reader +secret is supplied) and omits DATABASE_PASSWORD_READ_REPLICA / +DATABASE_URL_READ_REPLICA — the proxy mints the reader URL the same way. +Reader IAM only takes effect when the writer also uses IAM auth (the +proxy gates URL minting on IAM_TOKEN_DB_AUTH, which only the writer +sets). +*/}} +{{- define "litellm.serverEnv" -}} +{{- $root := .root -}} +{{- $component := .component -}} +- name: LITELLM_MASTER_KEY + valueFrom: + secretKeyRef: + name: {{ required "masterKey.secretName is required (the chart no longer accepts an inline master key)" $root.Values.masterKey.secretName }} + key: {{ $root.Values.masterKey.secretKey | default "master-key" }} +{{- if $component.logLevel }} +- name: LITELLM_LOG + value: {{ $component.logLevel | quote }} +{{- end }} +{{- with $root.Values.database.writer }} +- name: DATABASE_HOST + value: {{ required "database.writer.host is required" .host | quote }} +- name: DATABASE_PORT + value: {{ .port | default 5432 | quote }} +- name: DATABASE_USER + valueFrom: + secretKeyRef: + name: {{ required "database.writer.passwordSecret.name is required" .passwordSecret.name }} + key: {{ .passwordSecret.usernameKey | default "username" }} +- name: DATABASE_NAME + value: {{ required "database.writer.dbname is required" .dbname | quote }} +{{- if .schema }} +- name: DATABASE_SCHEMA + value: {{ .schema | quote }} +{{- end }} +{{- if .useIAMAuth }} +- name: IAM_TOKEN_DB_AUTH + value: "true" +{{- else }} +- name: DATABASE_PASSWORD + valueFrom: + secretKeyRef: + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.passwordKey | default "password" }} +- name: DATABASE_URL + value: "postgresql://$(DATABASE_USER):$(DATABASE_PASSWORD)@$(DATABASE_HOST):$(DATABASE_PORT)/$(DATABASE_NAME){{ if .schema }}?schema=$(DATABASE_SCHEMA){{ end }}" +{{- end }} +{{- end }} +{{- with $root.Values.database.reader }} +{{- if .host }} +{{- if and .useIAMAuth (not $root.Values.database.writer.useIAMAuth) }} +{{- fail "database.reader.useIAMAuth requires database.writer.useIAMAuth: true (the proxy gates IAM URL minting on IAM_TOKEN_DB_AUTH, which is only set by the writer)" }} +{{- end }} +{{- if .useIAMAuth }} +- name: DATABASE_HOST_READ_REPLICA + value: {{ .host | quote }} +- name: DATABASE_PORT_READ_REPLICA + value: {{ .port | default 5432 | quote }} +- name: DATABASE_NAME_READ_REPLICA + value: {{ required "database.reader.dbname is required when database.reader.host is set" .dbname | quote }} +{{- if .schema }} +- name: DATABASE_SCHEMA_READ_REPLICA + value: {{ .schema | quote }} +{{- end }} +{{- if .passwordSecret.name }} +- name: DATABASE_USER_READ_REPLICA + valueFrom: + secretKeyRef: + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.usernameKey | default "username" }} +{{- end }} +{{- else }} +{{- if not .passwordSecret.name }} +{{- fail "database.reader.passwordSecret.name is required when database.reader.host is set" }} +{{- end }} +- name: DATABASE_USER_READ_REPLICA + valueFrom: + secretKeyRef: + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.usernameKey | default "username" }} +- name: DATABASE_PASSWORD_READ_REPLICA + valueFrom: + secretKeyRef: + name: {{ .passwordSecret.name }} + key: {{ .passwordSecret.passwordKey | default "password" }} +- name: DATABASE_URL_READ_REPLICA + value: "postgresql://$(DATABASE_USER_READ_REPLICA):$(DATABASE_PASSWORD_READ_REPLICA)@{{ .host }}:{{ .port | default 5432 }}/{{ required "database.reader.dbname is required when database.reader.host is set" .dbname }}{{ if .schema }}?schema={{ .schema }}{{ end }}" +{{- end }} +{{- end }} +{{- end }} +{{- if $root.Values.redis.host }} +- name: REDIS_HOST + value: {{ $root.Values.redis.host | quote }} +- name: REDIS_PORT + value: {{ $root.Values.redis.port | quote }} +{{- if $root.Values.redis.passwordSecret.name }} +- name: REDIS_PASSWORD + valueFrom: + secretKeyRef: + name: {{ $root.Values.redis.passwordSecret.name }} + key: {{ $root.Values.redis.passwordSecret.passwordKey | default "password" }} +{{- end }} +{{- if $root.Values.redis.cluster }} +{{/* The proxy's Cache() reads REDIS_CLUSTER_NODES as JSON and constructs a + RedisClusterCache when it's set (litellm/caching/caching.py:169-192). + We seed with the single configured endpoint — the cluster client + discovers the remaining nodes from CLUSTER SLOTS at startup. */}} +- name: REDIS_CLUSTER_NODES + value: {{ printf "[{\"host\":%q,\"port\":%v}]" $root.Values.redis.host (int $root.Values.redis.port) | quote }} +{{- end }} +{{- end }} +{{- with $component.extraEnv }} +{{ toYaml . }} +{{- end }} +{{- end -}} + +{{/* +Renders `envFrom:` block for a component's `envConfigMaps` / `envSecrets` +lists. Each entry is a resource name; the chart wires the whole ConfigMap / +Secret into the container's env via configMapRef / secretRef. + +Invoke with just the component dict, e.g. `.Values.gateway`. Emits nothing +when both lists are empty so the container spec stays clean. +*/}} +{{- define "litellm.envFrom" -}} +{{- $component := . -}} +{{- if or $component.envConfigMaps $component.envSecrets }} +envFrom: +{{- range $component.envConfigMaps }} + - configMapRef: + name: {{ . }} +{{- end }} +{{- range $component.envSecrets }} + - secretRef: + name: {{ . }} +{{- end }} +{{- end }} +{{- end -}} diff --git a/helm/litellm/templates/backend/deployment.yaml b/helm/litellm/templates/backend/deployment.yaml new file mode 100644 index 00000000000..e761409f8c4 --- /dev/null +++ b/helm/litellm/templates/backend/deployment.yaml @@ -0,0 +1,60 @@ +{{- if .Values.backend.enabled }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ include "litellm.backend.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: backend +spec: + selector: + matchLabels: + {{- include "litellm.backend.selectorLabels" . | nindent 6 }} + template: + metadata: + {{- with .Values.backend.podAnnotations }} + annotations: + {{- toYaml . | nindent 8 }} + {{- end }} + labels: + {{- include "litellm.backend.selectorLabels" . | nindent 8 }} + spec: + serviceAccountName: {{ include "litellm.serviceAccountName" . }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: backend + image: "{{ .Values.backend.image.repository }}:{{ .Values.backend.image.tag | default .Chart.AppVersion }}" + imagePullPolicy: {{ .Values.backend.image.pullPolicy }} + ports: + - name: http + containerPort: 4001 + protocol: TCP + env: + {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.backend) | nindent 12 }} + {{- include "litellm.envFrom" .Values.backend | nindent 10 }} + {{- with .Values.backend.livenessProbe }} + livenessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.backend.readinessProbe }} + readinessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + resources: + {{- toYaml .Values.backend.resources | nindent 12 }} + {{- with .Values.backend.nodeSelector }} + nodeSelector: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.backend.affinity }} + affinity: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.backend.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/backend/hpa.yaml b/helm/litellm/templates/backend/hpa.yaml new file mode 100644 index 00000000000..d02f011d0bb --- /dev/null +++ b/helm/litellm/templates/backend/hpa.yaml @@ -0,0 +1,33 @@ +{{- if and .Values.backend.enabled .Values.backend.hpa.enabled }} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "litellm.backend.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: backend +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: {{ include "litellm.backend.fullname" . }} + minReplicas: {{ .Values.backend.hpa.minReplicas }} + maxReplicas: {{ .Values.backend.hpa.maxReplicas }} + metrics: + {{- if .Values.backend.hpa.targetCPUUtilizationPercentage }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ .Values.backend.hpa.targetCPUUtilizationPercentage }} + {{- end }} + {{- if .Values.backend.hpa.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ .Values.backend.hpa.targetMemoryUtilizationPercentage }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/backend/service.yaml b/helm/litellm/templates/backend/service.yaml new file mode 100644 index 00000000000..d480c654784 --- /dev/null +++ b/helm/litellm/templates/backend/service.yaml @@ -0,0 +1,18 @@ +{{- if .Values.backend.enabled }} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "litellm.backend.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: backend +spec: + type: {{ .Values.backend.service.type }} + ports: + - port: {{ .Values.backend.service.port }} + targetPort: http + protocol: TCP + name: http + selector: + {{- include "litellm.backend.selectorLabels" . | nindent 4 }} +{{- end }} diff --git a/helm/litellm/templates/gateway/configmap.yaml b/helm/litellm/templates/gateway/configmap.yaml new file mode 100644 index 00000000000..d262bf25b87 --- /dev/null +++ b/helm/litellm/templates/gateway/configmap.yaml @@ -0,0 +1,9 @@ +{{- if .Values.gateway.config.create }} +apiVersion: v1 +kind: ConfigMap +metadata: + name: {{ include "litellm.gateway.fullname" . }}-config +data: + config.yaml: | +{{ .Values.gateway.config.proxy_config | toYaml | indent 6 }} +{{- end }} diff --git a/helm/litellm/templates/gateway/deployment.yaml b/helm/litellm/templates/gateway/deployment.yaml new file mode 100644 index 00000000000..935d432342e --- /dev/null +++ b/helm/litellm/templates/gateway/deployment.yaml @@ -0,0 +1,83 @@ +{{- if .Values.gateway.enabled }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ include "litellm.gateway.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: gateway +spec: + selector: + matchLabels: + {{- include "litellm.gateway.selectorLabels" . | nindent 6 }} + template: + metadata: + annotations: + {{- if .Values.gateway.config.create }} + checksum/config: {{ include (print $.Template.BasePath "/gateway/configmap.yaml") . | sha256sum }} + {{- end }} + {{- with .Values.gateway.podAnnotations }} + {{- toYaml . | nindent 8 }} + {{- end }} + labels: + {{- include "litellm.gateway.selectorLabels" . | nindent 8 }} + spec: + serviceAccountName: {{ include "litellm.serviceAccountName" . }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: gateway + image: "{{ .Values.gateway.image.repository }}:{{ .Values.gateway.image.tag | default .Chart.AppVersion }}" + imagePullPolicy: {{ .Values.gateway.image.pullPolicy }} + ports: + - name: http + containerPort: 4000 + protocol: TCP + env: + {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.gateway) | nindent 12 }} + {{- if .Values.gateway.config.create }} + - name: CONFIG_FILE_PATH + value: /app/config/config.yaml + {{- end }} + {{- if .Values.gateway.numWorkers }} + - name: NUM_WORKERS + value: {{ .Values.gateway.numWorkers | quote }} + {{- end }} + {{- include "litellm.envFrom" .Values.gateway | nindent 10 }} + {{- if .Values.gateway.config.create }} + volumeMounts: + - name: gateway-config + mountPath: /app/config/config.yaml + subPath: config.yaml + {{- end }} + {{- with .Values.gateway.livenessProbe }} + livenessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.gateway.readinessProbe }} + readinessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + resources: + {{- toYaml .Values.gateway.resources | nindent 12 }} + {{- if .Values.gateway.config.create }} + volumes: + - name: gateway-config + configMap: + name: {{ include "litellm.gateway.fullname" . }}-config + {{- end }} + {{- with .Values.gateway.nodeSelector }} + nodeSelector: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.affinity }} + affinity: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.gateway.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/gateway/hpa.yaml b/helm/litellm/templates/gateway/hpa.yaml new file mode 100644 index 00000000000..27c4f05ba59 --- /dev/null +++ b/helm/litellm/templates/gateway/hpa.yaml @@ -0,0 +1,33 @@ +{{- if and .Values.gateway.enabled .Values.gateway.hpa.enabled }} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "litellm.gateway.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: gateway +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: {{ include "litellm.gateway.fullname" . }} + minReplicas: {{ .Values.gateway.hpa.minReplicas }} + maxReplicas: {{ .Values.gateway.hpa.maxReplicas }} + metrics: + {{- if .Values.gateway.hpa.targetCPUUtilizationPercentage }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ .Values.gateway.hpa.targetCPUUtilizationPercentage }} + {{- end }} + {{- if .Values.gateway.hpa.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ .Values.gateway.hpa.targetMemoryUtilizationPercentage }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/gateway/service.yaml b/helm/litellm/templates/gateway/service.yaml new file mode 100644 index 00000000000..03a4167a0ab --- /dev/null +++ b/helm/litellm/templates/gateway/service.yaml @@ -0,0 +1,18 @@ +{{- if .Values.gateway.enabled }} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "litellm.gateway.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: gateway +spec: + type: {{ .Values.gateway.service.type }} + ports: + - port: {{ .Values.gateway.service.port }} + targetPort: http + protocol: TCP + name: http + selector: + {{- include "litellm.gateway.selectorLabels" . | nindent 4 }} +{{- end }} diff --git a/helm/litellm/templates/ingress.yaml b/helm/litellm/templates/ingress.yaml new file mode 100644 index 00000000000..5e4ba780037 --- /dev/null +++ b/helm/litellm/templates/ingress.yaml @@ -0,0 +1,115 @@ +{{- if .Values.ingress.enabled -}} +{{- $gatewayName := include "litellm.gateway.fullname" . -}} +{{- $backendName := include "litellm.backend.fullname" . -}} +{{- $uiName := include "litellm.ui.fullname" . -}} +{{- $gatewayPort := .Values.gateway.service.port -}} +{{- $backendPort := .Values.backend.service.port -}} +{{- $uiPort := .Values.ui.service.port -}} +{{/* + Gateway data-plane prefixes — must mirror gateway/routes/allowlist.py. + Versioned paths are listed explicitly to avoid routing management routes + (e.g. /v1/access_group, /v2/key/info, /v1/tool/*, /v1/agents, /v1/workflows, + /v2/user/info, /v2/team/list, /v2/model/info, /v2/login, /v2/guardrails/*, + /v1/mcp/*) onto the gateway via a broad /v1 or /v2 prefix. +*/}} +{{- $gatewayPrefixes := list + "/v1/chat" "/chat" "/v1/completions" "/completions" "/v1/embeddings" "/embeddings" + "/v1/moderations" "/moderations" "/v1/audio" "/audio" "/v1/images" "/images" + "/v1/files" "/files" "/v1/batches" "/batches" "/v1/fine_tuning" "/fine_tuning" + "/v1/fine-tuning" "/fine-tuning" "/v1/responses" "/responses" "/v1/threads" "/threads" + "/v1/assistants" "/assistants" "/v1/vector_stores" "/vector_stores" "/v1/indexes" + "/v1/models" "/models" "/openai" "/engines" + "/v1/messages" "/messages" "/v1/skills" "/v1/a2a" + "/v1/rerank" "/v2/rerank" "/rerank" "/v1/ocr" "/ocr" "/v1/rag" "/rag" + "/v1/video" "/v1/videos" "/video" "/videos" "/v1/search" "/search" + "/v1/containers" "/containers" "/v1/evals" "/v1/memory" "/queue/chat" + "/v1beta" "/interactions" + "/anthropic" "/azure" "/azure_ai" "/aws" "/bedrock" "/cohere" "/gemini" "/google" + "/vertex_ai" "/vertex-ai" "/assemblyai" "/eu.assemblyai" "/langfuse" "/vllm" + "/mistral" "/groq" "/voyage" "/cursor" "/milvus" "/openai_passthrough" + "/toolset" + "/v1/realtime" "/realtime" + "/health" "/metrics" "/test" +-}} +apiVersion: networking.k8s.io/v1 +kind: Ingress +metadata: + name: {{ include "litellm.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + {{- with .Values.ingress.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + {{- with .Values.ingress.className }} + ingressClassName: {{ . | quote }} + {{- end }} + {{- with .Values.ingress.tls }} + tls: + {{- toYaml . | nindent 4 }} + {{- end }} + rules: + - {{- with .Values.ingress.host }} + host: {{ . | quote }} + {{- end }} + http: + paths: + # --- UI (Next.js static export) --- + - path: / + pathType: Exact + backend: + service: + name: {{ $uiName }} + port: + number: {{ $uiPort }} + - path: /favicon.ico + pathType: Exact + backend: + service: + name: {{ $uiName }} + port: + number: {{ $uiPort }} + - path: /litellm-asset-prefix + pathType: Prefix + backend: + service: + name: {{ $uiName }} + port: + number: {{ $uiPort }} + - path: /_next + pathType: Prefix + backend: + service: + name: {{ $uiName }} + port: + number: {{ $uiPort }} + # /ui/* is where the Next.js SPA serves its login + dashboard + # routes (e.g. /ui/login). Without this, /ui/* falls into the + # catch-all → backend → 404. + - path: /ui + pathType: Prefix + backend: + service: + name: {{ $uiName }} + port: + number: {{ $uiPort }} + # --- Gateway data plane --- + {{- range $gatewayPrefixes }} + - path: {{ . }} + pathType: Prefix + backend: + service: + name: {{ $gatewayName }} + port: + number: {{ $gatewayPort }} + {{- end }} + # --- Catch-all → backend (management API: /key/*, /user/*, /team/*, ...) --- + - path: / + pathType: Prefix + backend: + service: + name: {{ $backendName }} + port: + number: {{ $backendPort }} +{{- end }} diff --git a/helm/litellm/templates/migrations-job.yaml b/helm/litellm/templates/migrations-job.yaml new file mode 100644 index 00000000000..c5e000bfd02 --- /dev/null +++ b/helm/litellm/templates/migrations-job.yaml @@ -0,0 +1,48 @@ +{{- if .Values.migrationJob.enabled -}} +# Post-install / post-upgrade hook that runs `prisma migrate deploy` against +# the writer database. Required because the gateway and backend both spin up +# Prisma at startup and assume the LiteLLM schema (LiteLLM_Config, +# LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...) already exists. +apiVersion: batch/v1 +kind: Job +metadata: + name: {{ include "litellm.fullname" . }}-migrations + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: migrations + annotations: + helm.sh/hook: post-install,post-upgrade + helm.sh/hook-delete-policy: before-hook-creation + helm.sh/hook-weight: "0" +spec: + backoffLimit: {{ .Values.migrationJob.backoffLimit }} + ttlSecondsAfterFinished: {{ .Values.migrationJob.ttlSecondsAfterFinished }} + template: + metadata: + labels: + {{- include "litellm.commonLabels" . | nindent 8 }} + app.kubernetes.io/component: migrations + spec: + restartPolicy: Never + serviceAccountName: {{ include "litellm.serviceAccountName" . }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: prisma-migrations + image: "{{ .Values.backend.image.repository }}:{{ .Values.backend.image.tag | default .Chart.AppVersion }}" + imagePullPolicy: {{ .Values.backend.image.pullPolicy }} + workingDir: /app + command: ["python", "litellm/proxy/prisma_migration.py"] + env: + {{- include "litellm.serverEnv" (dict "root" $ "component" .Values.backend) | nindent 12 }} + # Force the migration to run even if a prior pod set + # DISABLE_SCHEMA_UPDATE=true to skip auto-migrations at runtime. + - name: DISABLE_SCHEMA_UPDATE + value: "false" + {{- with .Values.migrationJob.resources }} + resources: + {{- toYaml . | nindent 12 }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/serviceaccount.yaml b/helm/litellm/templates/serviceaccount.yaml new file mode 100644 index 00000000000..3c998448ae5 --- /dev/null +++ b/helm/litellm/templates/serviceaccount.yaml @@ -0,0 +1,13 @@ +{{- if .Values.serviceAccount.create -}} +apiVersion: v1 +kind: ServiceAccount +metadata: + name: {{ include "litellm.serviceAccountName" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + {{- with .Values.serviceAccount.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +automountServiceAccountToken: {{ .Values.serviceAccount.automount }} +{{- end }} diff --git a/helm/litellm/templates/ui/deployment.yaml b/helm/litellm/templates/ui/deployment.yaml new file mode 100644 index 00000000000..549bf61a0dd --- /dev/null +++ b/helm/litellm/templates/ui/deployment.yaml @@ -0,0 +1,70 @@ +{{- if .Values.ui.enabled }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ include "litellm.ui.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: ui +spec: + selector: + matchLabels: + {{- include "litellm.ui.selectorLabels" . | nindent 6 }} + template: + metadata: + {{- with .Values.ui.podAnnotations }} + annotations: + {{- toYaml . | nindent 8 }} + {{- end }} + labels: + {{- include "litellm.ui.selectorLabels" . | nindent 8 }} + spec: + serviceAccountName: {{ include "litellm.serviceAccountName" . }} + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: ui + image: "{{ .Values.ui.image.repository }}:{{ .Values.ui.image.tag | default .Chart.AppVersion }}" + imagePullPolicy: {{ .Values.ui.image.pullPolicy }} + ports: + - name: http + containerPort: 3000 + protocol: TCP + env: + {{- if .Values.ui.logLevel }} + - name: LITELLM_LOG + value: {{ .Values.ui.logLevel | quote }} + {{- end }} + {{- if .Values.ui.backendUrl }} + - name: LITELLM_BACKEND_URL + value: {{ .Values.ui.backendUrl | quote }} + {{- end }} + {{- with .Values.ui.extraEnv }} + {{- toYaml . | nindent 12 }} + {{- end }} + {{- include "litellm.envFrom" .Values.ui | nindent 10 }} + {{- with .Values.ui.livenessProbe }} + livenessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.ui.readinessProbe }} + readinessProbe: + {{- toYaml . | nindent 12 }} + {{- end }} + resources: + {{- toYaml .Values.ui.resources | nindent 12 }} + {{- with .Values.ui.nodeSelector }} + nodeSelector: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.ui.affinity }} + affinity: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.ui.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/ui/hpa.yaml b/helm/litellm/templates/ui/hpa.yaml new file mode 100644 index 00000000000..b43eda5ac4a --- /dev/null +++ b/helm/litellm/templates/ui/hpa.yaml @@ -0,0 +1,33 @@ +{{- if and .Values.ui.enabled .Values.ui.hpa.enabled }} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "litellm.ui.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: ui +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: {{ include "litellm.ui.fullname" . }} + minReplicas: {{ .Values.ui.hpa.minReplicas }} + maxReplicas: {{ .Values.ui.hpa.maxReplicas }} + metrics: + {{- if .Values.ui.hpa.targetCPUUtilizationPercentage }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ .Values.ui.hpa.targetCPUUtilizationPercentage }} + {{- end }} + {{- if .Values.ui.hpa.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ .Values.ui.hpa.targetMemoryUtilizationPercentage }} + {{- end }} +{{- end }} diff --git a/helm/litellm/templates/ui/service.yaml b/helm/litellm/templates/ui/service.yaml new file mode 100644 index 00000000000..52b539fa00c --- /dev/null +++ b/helm/litellm/templates/ui/service.yaml @@ -0,0 +1,18 @@ +{{- if .Values.ui.enabled }} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "litellm.ui.fullname" . }} + labels: + {{- include "litellm.commonLabels" . | nindent 4 }} + app.kubernetes.io/component: ui +spec: + type: {{ .Values.ui.service.type }} + ports: + - port: {{ .Values.ui.service.port }} + targetPort: http + protocol: TCP + name: http + selector: + {{- include "litellm.ui.selectorLabels" . | nindent 4 }} +{{- end }} diff --git a/helm/litellm/values.yaml b/helm/litellm/values.yaml new file mode 100644 index 00000000000..64011829fff --- /dev/null +++ b/helm/litellm/values.yaml @@ -0,0 +1,211 @@ +# LiteLLM helm chart values + +nameOverride: "" +fullnameOverride: "" + +imagePullSecrets: [] + +# Optional Ingress wiring the three component Services behind a single L7 +# entrypoint. Required when serving the static UI bundle over the network. +ingress: + enabled: false + className: "" + annotations: {} + host: "" # optional; if set, becomes the rule's host + tls: [] + +# Shared ServiceAccount used by all three component Deployments. Set +# `create: true` to have the chart provision it (e.g. when wiring an EKS +# Pod Identity association by SA name). Set `name` to use an existing SA +# (chart-created or out-of-band). When both are empty / false, pods run +# with the namespace's `default` SA. +serviceAccount: + create: false + automount: true + annotations: {} + name: "" + +# Post-install / post-upgrade Helm hook that runs `prisma migrate deploy` +# against the writer database, creating the LiteLLM schema (tables that +# gateway + backend assume exist at startup: LiteLLM_Config, +# LiteLLM_VerificationToken, LiteLLM_SpendLogs, ...). Disable if your +# pipeline runs migrations out-of-band. +migrationJob: + enabled: true + backoffLimit: 4 + ttlSecondsAfterFinished: 120 + resources: {} + +# Required: a master key used by gateway + backend to mint/verify proxy tokens. +# Must reference an existing Secret. +masterKey: + secretName: litellm-master-key-secret # name of a Secret containing the master key + secretKey: master-key + +# External Postgres connection. +database: + writer: + host: "" + port: 5432 + dbname: "" + schema: "" + useIAMAuth: false + passwordSecret: + name: litellm-writer-secret + usernameKey: username + passwordKey: password + + # Optional read-replica routing. When `reader.host` is set, the proxy routes + # reads (find_*, count, group_by, query_raw/_first) to this endpoint while + # writes stay on the writer. Leave `reader.host` empty to disable. + reader: + host: "" + port: 5432 + dbname: "" + schema: "" + useIAMAuth: false + passwordSecret: + name: litellm-reader-secret + usernameKey: username + passwordKey: password + +# Optional Redis (caching, rate limiting). Leave host empty to disable. +# +# Set `cluster: true` for Redis Cluster mode (e.g. AWS ElastiCache Cluster, +# self-hosted Redis Cluster). The chart emits REDIS_CLUSTER_NODES from +# `host` / `port` as the single seed; the cluster client discovers the +# remaining nodes from CLUSTER SLOTS at startup. +redis: + cluster: false + host: "" + port: 6379 + passwordSecret: + name: "" # Leave empty for auth-less Redis + passwordKey: password + +# ---------- gateway (LLM data plane) ---------- +gateway: + enabled: true + logLevel: INFO + # Number of uvicorn worker processes per gateway pod. Sets NUM_WORKERS, + # consumed by the gateway image entrypoint. Default is 1. + numWorkers: 1 + extraEnv: [] # Add extra environment variables to the gateway + envConfigMaps: [] # Add extra environment variables to the gateway from config maps + envSecrets: [] # Add extra environment variables to the gateway from secrets + config: + create: true + proxy_config: {} + image: + repository: ghcr.io/berriai/litellm-gateway + tag: "" # defaults to .Chart.AppVersion + pullPolicy: IfNotPresent + service: + type: ClusterIP + port: 4000 + resources: + requests: + cpu: "1" + memory: 4Gi + limits: + cpu: "2" + memory: 4Gi + livenessProbe: + httpGet: { path: /health/liveliness, port: http } + initialDelaySeconds: 10 + periodSeconds: 15 + readinessProbe: + httpGet: { path: /health/readiness, port: http } + initialDelaySeconds: 5 + periodSeconds: 10 + hpa: + enabled: true + minReplicas: 1 + maxReplicas: 10 + targetCPUUtilizationPercentage: 70 + targetMemoryUtilizationPercentage: 80 + podAnnotations: {} + nodeSelector: {} + tolerations: [] + affinity: {} + +# ---------- backend (UI / management API) ---------- +backend: + enabled: true + logLevel: INFO + extraEnv: [] + envConfigMaps: [] + envSecrets: [] + image: + repository: ghcr.io/berriai/litellm-backend + tag: "" + pullPolicy: IfNotPresent + service: + type: ClusterIP + port: 4001 + resources: + requests: + cpu: "1" + memory: 4Gi + limits: + cpu: "2" + memory: 4Gi + livenessProbe: + httpGet: { path: /health/liveliness, port: http } + initialDelaySeconds: 10 + periodSeconds: 15 + readinessProbe: + httpGet: { path: /health/readiness, port: http } + initialDelaySeconds: 5 + periodSeconds: 10 + hpa: + enabled: true + minReplicas: 1 + maxReplicas: 4 + targetCPUUtilizationPercentage: 70 + podAnnotations: {} + nodeSelector: {} + tolerations: [] + affinity: {} + +# ---------- ui (Next.js static dashboard) ---------- +ui: + enabled: true + logLevel: INFO + extraEnv: [] + envConfigMaps: [] + envSecrets: [] + image: + repository: ghcr.io/berriai/litellm-ui + tag: "" + pullPolicy: IfNotPresent + service: + type: ClusterIP + port: 3000 + # The dashboard expects to know where to reach the backend API. Set this to + # the externally-routable URL (typically the ingress host + /api or similar). + backendUrl: "" + resources: + requests: + cpu: 500m + memory: 500Mi + limits: + cpu: "1" + memory: 1Gi + livenessProbe: + httpGet: { path: /, port: http } + initialDelaySeconds: 5 + periodSeconds: 20 + readinessProbe: + httpGet: { path: /, port: http } + initialDelaySeconds: 2 + periodSeconds: 10 + hpa: + enabled: false + minReplicas: 1 + maxReplicas: 3 + targetCPUUtilizationPercentage: 80 + podAnnotations: {} + nodeSelector: {} + tolerations: [] + affinity: {} diff --git a/litellm/proxy/auth/rds_iam_token.py b/litellm/proxy/auth/rds_iam_token.py index 053cdb91f17..b0b156d786e 100644 --- a/litellm/proxy/auth/rds_iam_token.py +++ b/litellm/proxy/auth/rds_iam_token.py @@ -159,6 +159,100 @@ def init_rds_client( return client +def _build_iam_db_url( + db_host: str, + db_port: str, + db_user: str, + db_name: str, + db_schema: Optional[str], +) -> str: + token = generate_iam_auth_token(db_host=db_host, db_port=db_port, db_user=db_user) + url = f"postgresql://{db_user}:{token}@{db_host}:{db_port}/{db_name}" + if db_schema: + url += f"?schema={db_schema}" + return url + + +def init_iam_db_url_from_env() -> bool: + """Assemble ``DATABASE_URL`` (+ ``DATABASE_URL_READ_REPLICA``) from RDS IAM + env vars before Prisma initializes. + + The standard CLI flow (``litellm/proxy/proxy_cli.py``) does this just + before importing the FastAPI app. Componentized entrypoints + (``gateway/main.py``, ``backend/main.py``) bypass that CLI and uvicorn the + app directly, so they must call this themselves before importing + ``litellm.proxy.proxy_server`` — otherwise Prisma boots with the + placeholder URL and every DB-backed endpoint returns + "Database not connected". + + Reads: + IAM_TOKEN_DB_AUTH (required to enable; case-insensitive truthy). + Writer: DATABASE_HOST, DATABASE_PORT (default 5432), DATABASE_USER, + DATABASE_NAME, DATABASE_SCHEMA (optional). + Reader (optional, only if DATABASE_HOST_READ_REPLICA is set): + DATABASE_HOST_READ_REPLICA, DATABASE_PORT_READ_REPLICA + (default 5432), DATABASE_USER_READ_REPLICA (defaults to + DATABASE_USER), DATABASE_NAME_READ_REPLICA (defaults to + DATABASE_NAME), DATABASE_SCHEMA_READ_REPLICA (defaults to + DATABASE_SCHEMA). + + Sets: + DATABASE_URL and (when reader env vars are present) + DATABASE_URL_READ_REPLICA. + + Returns: + True if IAM auth was enabled and at least the writer URL was + assembled, False if IAM auth was disabled. + """ + from litellm.secret_managers.main import get_secret_bool + + if not get_secret_bool("IAM_TOKEN_DB_AUTH"): + return False + + # Writer — required. + db_host = os.getenv("DATABASE_HOST") + db_port = os.getenv("DATABASE_PORT", "5432") + db_user = os.getenv("DATABASE_USER") + db_name = os.getenv("DATABASE_NAME") + db_schema = os.getenv("DATABASE_SCHEMA") + + if not (db_host and db_user and db_name): + raise RuntimeError( + "IAM_TOKEN_DB_AUTH is set but DATABASE_HOST / DATABASE_USER / " + "DATABASE_NAME are required to assemble DATABASE_URL." + ) + + os.environ["DATABASE_URL"] = _build_iam_db_url( + db_host=db_host, + db_port=db_port, + db_user=db_user, + db_name=db_name, + db_schema=db_schema, + ) + + # Reader — optional. Only assembled when a reader host is configured; + # remaining fields fall back to the writer's values so callers that share + # one DB user / DB name across writer + reader don't have to duplicate env + # vars. (Reader IAM-refresh at runtime is handled in + # ``litellm/proxy/utils.py`` via ``parse_iam_endpoint_from_url``.) + reader_host = os.getenv("DATABASE_HOST_READ_REPLICA") + if reader_host: + reader_port = os.getenv("DATABASE_PORT_READ_REPLICA", "5432") + reader_user = os.getenv("DATABASE_USER_READ_REPLICA", db_user) + reader_name = os.getenv("DATABASE_NAME_READ_REPLICA", db_name) + reader_schema = os.getenv("DATABASE_SCHEMA_READ_REPLICA", db_schema) + + os.environ["DATABASE_URL_READ_REPLICA"] = _build_iam_db_url( + db_host=reader_host, + db_port=reader_port, + db_user=reader_user, + db_name=reader_name, + db_schema=reader_schema, + ) + + return True + + def generate_iam_auth_token( db_host, db_port, db_user, client: Optional[Any] = None ) -> str: diff --git a/litellm/proxy/proxy_cli.py b/litellm/proxy/proxy_cli.py index 5fc8c44b2d8..425a0f71e97 100644 --- a/litellm/proxy/proxy_cli.py +++ b/litellm/proxy/proxy_cli.py @@ -811,30 +811,10 @@ def run_server( # noqa: PLR0915 ### GET DB TOKEN FOR IAM AUTH ### if iam_token_db_auth or get_secret_bool("IAM_TOKEN_DB_AUTH"): - from litellm.proxy.auth.rds_iam_token import generate_iam_auth_token + from litellm.proxy.auth.rds_iam_token import init_iam_db_url_from_env - db_host = os.getenv("DATABASE_HOST") - # Default to the Postgres standard port. Without a default, - # `db_port=None` flows into `boto.generate_db_auth_token(Port=None)` - # and botocore stringifies it to `"None"` while building the - # presigned URL, which then blows up with `ValueError: Port could - # not be cast to integer value as 'None'` during signing. - db_port = os.getenv("DATABASE_PORT", "5432") - db_user = os.getenv("DATABASE_USER") - db_name = os.getenv("DATABASE_NAME") - db_schema = os.getenv("DATABASE_SCHEMA") - - token = generate_iam_auth_token( - db_host=db_host, db_port=db_port, db_user=db_user - ) - - # print(f"token: {token}") - _db_url = f"postgresql://{db_user}:{token}@{db_host}:{db_port}/{db_name}" - if db_schema: - _db_url += f"?schema={db_schema}" - - os.environ["DATABASE_URL"] = _db_url os.environ["IAM_TOKEN_DB_AUTH"] = "True" + init_iam_db_url_from_env() ### DECRYPT ENV VAR ### diff --git a/tests/test_litellm/proxy/test_component_allowlists.py b/tests/test_litellm/proxy/test_component_allowlists.py new file mode 100644 index 00000000000..4882e1f0583 --- /dev/null +++ b/tests/test_litellm/proxy/test_component_allowlists.py @@ -0,0 +1,68 @@ +"""Coverage test for the gateway / backend component allowlists. + +The componentization scaffold splits the proxy FastAPI app into two runtime +components by trimming the route table at import time: + + gateway.main -> only paths matched by gateway/routes/allowlist.py + backend.main -> only paths matched by backend/routes/allowlist.py + +If either allowlist drops a path that was reachable on the monolithic app, +clients hitting that path on the corresponding pod get a 404. This test +guarantees that the union of the two trimmed route sets equals the full set +of routes on the proxy app — i.e. no endpoint is dropped on the floor. + +The test reproduces the same predicate that ``gateway/main.py`` and +``backend/main.py`` use, without importing them (importing those modules +mutates the global ``app.router.routes`` and would corrupt the snapshot). +""" + +import os +import sys + +from fastapi.routing import Mount + +# gateway/ and backend/ live at the repo root, not inside litellm/. +_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..", "..")) +if _REPO_ROOT not in sys.path: + sys.path.insert(0, _REPO_ROOT) + +from backend.routes.allowlist import BACKEND_EXACT_PATHS, BACKEND_PATH_PREFIXES +from gateway.routes.allowlist import GATEWAY_EXACT_PATHS, GATEWAY_PATH_PREFIXES +from litellm.proxy.proxy_server import app + + +def _component_paths(routes, exact_paths, path_prefixes) -> set[str]: + """Reproduce ``gateway.main._is_gateway_route`` / ``backend.main._is_backend_route``.""" + out: set[str] = set() + for route in routes: + if isinstance(route, Mount): + continue + path = getattr(route, "path", None) + if path is None: + continue + if path in exact_paths or any(path.startswith(p) for p in path_prefixes): + out.add(path) + return out + + +def test_gateway_plus_backend_covers_full_app(): + """Every route on the proxy app must be served by gateway or backend.""" + all_paths = { + getattr(r, "path") + for r in app.router.routes + if not isinstance(r, Mount) and getattr(r, "path", None) is not None + } + gateway_paths = _component_paths( + app.router.routes, GATEWAY_EXACT_PATHS, GATEWAY_PATH_PREFIXES + ) + backend_paths = _component_paths( + app.router.routes, BACKEND_EXACT_PATHS, BACKEND_PATH_PREFIXES + ) + + uncovered = all_paths - (gateway_paths | backend_paths) + + assert not uncovered, ( + f"{len(uncovered)} route(s) are not exposed on either component. " + f"Update gateway/routes/allowlist.py or backend/routes/allowlist.py to cover:\n " + + "\n ".join(sorted(uncovered)) + ) diff --git a/ui/Dockerfile b/ui/Dockerfile new file mode 100644 index 00000000000..b75c4d0a0c6 --- /dev/null +++ b/ui/Dockerfile @@ -0,0 +1,42 @@ +# syntax=docker/dockerfile:1.7 + +# UI container — Next.js static export served by nginx. + +ARG NODE_VERSION=20.18-alpine3.20 +ARG NGINX_VERSION=1.27-alpine + +# ---------- builder ---------- +FROM node:${NODE_VERSION} AS builder + +ENV NEXT_TELEMETRY_DISABLED=1 \ + npm_config_fund=false \ + npm_config_audit=false + +WORKDIR /app + +# Layer the lockfile-only install above the source copy so source-only +# edits don't bust the install cache. +COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./ +RUN --mount=type=cache,target=/root/.npm \ + npm ci --prefer-offline + +COPY ui/litellm-dashboard/ ./ +RUN npm run build + +# ---------- runtime ---------- +FROM nginx:${NGINX_VERSION} AS runtime + +# Drop the upstream default :80 server; we own the config. +RUN rm -f /etc/nginx/conf.d/default.conf + +# Static export → web root. +COPY --from=builder /app/out /usr/share/nginx/html + +# Routing rules — see ui/nginx.conf for the full description. +COPY ui/nginx.conf /etc/nginx/nginx.conf + +EXPOSE 3000/tcp + +# nginx as PID 1 in foreground; respects SIGTERM out of the box, so +# no tini/dumb-init wrapper needed. +CMD ["nginx", "-g", "daemon off;"] diff --git a/ui/nginx.conf b/ui/nginx.conf new file mode 100644 index 00000000000..ac9fc7443f1 --- /dev/null +++ b/ui/nginx.conf @@ -0,0 +1,83 @@ +worker_processes auto; +events { worker_connections 1024; } + +http { + include /etc/nginx/mime.types; + default_type application/octet-stream; + sendfile on; + tcp_nopush on; + keepalive_timeout 65; + + gzip on; + gzip_comp_level 4; + gzip_min_length 1024; + gzip_proxied any; + gzip_types + application/javascript + application/json + text/css + text/html + image/svg+xml + font/woff + font/woff2; + + server { + listen 3000 default_server; + server_name _; + root /usr/share/nginx/html; + + # next.config.mjs sets assetPrefix=/litellm-asset-prefix, which makes + # the built HTML reference /litellm-asset-prefix/_next/... — but the + # static export only emits files under /_next/. Map the prefix to + # the real tree at request time instead of duplicating the directory + # at build time. NB: alias rewrites the location prefix, so + # /litellm-asset-prefix/_next/foo.js → /usr/share/nginx/html/_next/foo.js. + location /litellm-asset-prefix/_next/ { + alias /usr/share/nginx/html/_next/; + expires 1y; + add_header Cache-Control "public, immutable"; + } + + # Content-hashed asset bundles — cache forever. + location /_next/ { + try_files $uri =404; + expires 1y; + add_header Cache-Control "public, immutable"; + } + location /assets/ { + try_files $uri =404; + expires 1y; + add_header Cache-Control "public, immutable"; + } + location = /favicon.ico { + try_files $uri =404; + expires 1d; + } + + # Probe target — doesn't depend on disk. + location = /healthz { default_type text/plain; return 200 "ok\n"; } + + # /ui[/] — the dashboard's JS hardcodes URLs under this prefix + # (router.replace("/ui"), buildLoginUrlWithReturn("/ui/login"), ...). + # Mirror what FastAPI StaticFiles(mount="/ui") did in the monolithic + # proxy_server: serve /ui/ from out/.html, with App + # Router-aware fallback (out//index.html) and a final SPA + # fallback to out/index.html for client-side routes. + location = /ui { try_files /index.html =404; } + location = /ui/ { try_files /index.html =404; } + location ~ ^/ui/(.+)$ { + try_files /$1.html /$1/index.html /index.html =404; + } + + # `/` is handy for direct-debug port-forwards. + location = / { try_files /index.html =404; } + + # Anything else (API calls etc.) returns 404 from the UI's + # perspective. A reverse proxy in front of this image routes the + # API surface (/v1, /key, /.well-known/litellm-ui-config, ...) to + # gateway/backend before requests get here; if something slips + # through, fall through to a 404 instead of accidentally serving + # HTML and confusing a JSON-expecting caller. + location / { return 404; } + } +}