mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(terraform): one-click + zero-config deploy for AWS and GCP stacks
Make the LiteLLM Terraform stacks publishable and 1-click deployable: - Zero-config trial: a bare `terraform apply` in examples/default now works with no tfvars. AWS auto-picks the first two AZs and defaults region/tenant/env + HTTP-only trial LB; GCP infers the project from the active gcloud/ADC config and defaults the trial LB to HTTP-only. - GCP image pulls: auto-create an Artifact Registry remote repository that proxies ghcr.io (Cloud Run rejects ghcr.io directly), wire it as the default image registry, and grant the serverless agent read on it — so the default images pull with no manual mirroring. Guarded by a precondition that fails fast if images would still resolve to ghcr.io. - 1-click launch: GCP "Open in Cloud Shell" button + guided tutorial, AWS CloudShell button + checksum-verified deploy.sh helper + tutorial. Buttons added to the main and terraform READMEs. - Registry publishing: gated GitHub Actions workflow that mirrors each module subtree to its terraform-<provider>-litellm repo and tags it, plus docs. - Correctness fixes from review: pin the AWS backend listener to :4001 on the no-config path (matches the health check), add a name-length precondition for AWS ELB/target-group 32-char limits, wire GCP labels onto Cloud Run / Cloud SQL / Memorystore, and reconcile stale README image tags / migration command / TLS notes. https://claude.ai/code/session_01KprYNJjvub4vNYr4d2xCQ6
This commit is contained in:
parent
b8d6ff7eb3
commit
4e92c4f741
23 changed files with 779 additions and 86 deletions
98
.github/workflows/terraform-modules-publish.yml
vendored
Normal file
98
.github/workflows/terraform-modules-publish.yml
vendored
Normal file
|
|
@ -0,0 +1,98 @@
|
|||
# Publishes the LiteLLM Terraform modules to the public Terraform Registry.
|
||||
#
|
||||
# The registry only indexes a module that lives at the ROOT of a repo named
|
||||
# `terraform-<PROVIDER>-<NAME>`, so the two modules in this monorepo
|
||||
# (terraform/litellm/aws, terraform/litellm/gcp) can't be published in place.
|
||||
# This workflow mirrors each module subtree out to its own dedicated repo and
|
||||
# tags it, so the registry picks up the new version.
|
||||
#
|
||||
# ── One-time setup ───────────────────────────────────────────────────────────
|
||||
# 1. Create two empty GitHub repos:
|
||||
# BerriAI/terraform-aws-litellm
|
||||
# BerriAI/terraform-google-litellm
|
||||
# 2. Connect each to the Terraform Registry (registry.terraform.io → Publish →
|
||||
# Module) once; subsequent tags are ingested automatically.
|
||||
# 3. Add a repo secret `TERRAFORM_REGISTRY_SYNC_TOKEN` — a PAT (or fine-grained
|
||||
# token) with `contents:write` on both mirror repos. Without it this
|
||||
# workflow no-ops.
|
||||
#
|
||||
# ── Usage ────────────────────────────────────────────────────────────────────
|
||||
# Actions → "Publish Terraform modules" → Run workflow, enter the version
|
||||
# (e.g. v1.86.0). Leave dry_run checked first to see what it would push.
|
||||
name: Publish Terraform modules
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: "Module version tag to publish (semver, e.g. v1.86.0)"
|
||||
required: true
|
||||
type: string
|
||||
dry_run:
|
||||
description: "Print actions without pushing"
|
||||
required: false
|
||||
type: boolean
|
||||
default: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- prefix: terraform/litellm/aws
|
||||
repo: terraform-aws-litellm
|
||||
- prefix: terraform/litellm/gcp
|
||||
repo: terraform-google-litellm
|
||||
steps:
|
||||
- name: Validate version input
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if ! printf '%s' "${{ inputs.version }}" | grep -Eq '^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z.-]+)?$'; then
|
||||
echo "version must be semver with a leading v (e.g. v1.86.0)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Checkout (full history for subtree split)
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Mirror ${{ matrix.prefix }} → BerriAI/${{ matrix.repo }}
|
||||
env:
|
||||
TOKEN: ${{ secrets.TERRAFORM_REGISTRY_SYNC_TOKEN }}
|
||||
PREFIX: ${{ matrix.prefix }}
|
||||
REPO: ${{ matrix.repo }}
|
||||
VERSION: ${{ inputs.version }}
|
||||
DRY_RUN: ${{ inputs.dry_run }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${TOKEN:-}" ]; then
|
||||
echo "TERRAFORM_REGISTRY_SYNC_TOKEN not set — skipping (see workflow header for setup)."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
git config user.name "litellm-release-bot"
|
||||
git config user.email "release-bot@litellm.ai"
|
||||
|
||||
# Split the module subdirectory into a standalone commit graph whose
|
||||
# root is the module itself (what the registry expects).
|
||||
split_sha="$(git subtree split --prefix="$PREFIX" HEAD)"
|
||||
echo "subtree split for $PREFIX -> $split_sha"
|
||||
|
||||
remote="https://x-access-token:${TOKEN}@github.com/BerriAI/${REPO}.git"
|
||||
|
||||
if [ "$DRY_RUN" = "true" ]; then
|
||||
echo "[dry-run] would push $split_sha to ${REPO}:main and tag $VERSION"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Publish the split as the mirror's main branch + the version tag.
|
||||
git push --force "$remote" "${split_sha}:refs/heads/main"
|
||||
git tag -f "$VERSION" "$split_sha"
|
||||
git push --force "$remote" "refs/tags/${VERSION}"
|
||||
echo "Published BerriAI/${REPO}@${VERSION}"
|
||||
|
|
@ -10,7 +10,11 @@
|
|||
<a href="https://railway.com/deploy/RhvhdC?referralCode=7mRv9K&utm_medium=integration&utm_source=template&utm_campaign=generic">
|
||||
<img src="https://railway.com/button.svg" alt="Deploy on Railway">
|
||||
</a>
|
||||
<a href="https://shell.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https://github.com/BerriAI/litellm&cloudshell_workspace=terraform/litellm/gcp/examples/default&cloudshell_tutorial=tutorial.md" target="_blank" rel="nofollow"><img src="https://gstatic.com/cloudssh/images/open-btn.svg" alt="Deploy on GCP (Cloud Shell)"></a>
|
||||
<a href="https://console.aws.amazon.com/cloudshell/home" target="_blank" rel="nofollow"><img src="https://img.shields.io/badge/Deploy%20on-AWS-FF9900?logo=amazonaws&logoColor=white" alt="Deploy on AWS (CloudShell)"></a>
|
||||
</p>
|
||||
<p align="center"><sub>One-click cloud deploys (Terraform) →
|
||||
<a href="https://github.com/BerriAI/litellm/tree/main/terraform/litellm">terraform/litellm</a></sub></p>
|
||||
</p>
|
||||
<h4 align="center"><a href="https://docs.litellm.ai/docs/simple_proxy" target="_blank">LiteLLM Proxy Server (AI Gateway)</a> | <a href="https://docs.litellm.ai/docs/enterprise#hosted-litellm-proxy" target="_blank"> Hosted Proxy</a> | <a href="https://litellm.ai/enterprise"target="_blank">Enterprise Tier</a> | <a href="https://www.litellm.ai/ai-gateway" target="_blank">Website</a></h4>
|
||||
<h4 align="center">
|
||||
|
|
|
|||
|
|
@ -13,10 +13,18 @@ one-command deploy path. To embed a stack in your own config, call the module
|
|||
by source:
|
||||
|
||||
```hcl
|
||||
# Direct from this monorepo (works today, no registry needed):
|
||||
module "litellm" {
|
||||
source = "github.com/BerriAI/litellm//terraform/litellm/aws?ref=<tag>"
|
||||
# ... inputs ...
|
||||
}
|
||||
|
||||
# Or, once published (see "Publishing to the Terraform Registry" below):
|
||||
module "litellm" {
|
||||
source = "BerriAI/litellm/aws" # registry shorthand
|
||||
version = "~> 1.86"
|
||||
# ... inputs ...
|
||||
}
|
||||
```
|
||||
|
||||
| Stack | Compute | Database (writer + reader) | Cache | Object store | Public entrypoint |
|
||||
|
|
@ -30,15 +38,52 @@ Both stacks support a typed `proxy_config` input (mirrors `helm/litellm`'s
|
|||
`gateway.config.proxy_config`) and per-component extra env vars /
|
||||
secret-manager refs.
|
||||
|
||||
## One-click deploy
|
||||
|
||||
The `examples/default/` roots are **zero-config trial deploys**: a bare
|
||||
`terraform apply` with no tfvars brings up a working (HTTP-only) instance —
|
||||
sensible region/zone defaults, an auto-generated master key, and (on GCP) an
|
||||
auto-created Artifact Registry proxy so Cloud Run can pull the images. Launch
|
||||
straight from your browser:
|
||||
|
||||
### GCP — Open in Cloud Shell
|
||||
|
||||
[](https://shell.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https://github.com/BerriAI/litellm&cloudshell_workspace=terraform/litellm/gcp/examples/default&cloudshell_tutorial=tutorial.md)
|
||||
|
||||
Opens Cloud Shell (Terraform is pre-installed), clones the repo, and starts a
|
||||
guided walkthrough that enables the APIs and runs `terraform apply` against
|
||||
your active project.
|
||||
|
||||
### AWS — Open in CloudShell
|
||||
|
||||
[](https://console.aws.amazon.com/cloudshell/home)
|
||||
|
||||
CloudShell already has your AWS credentials. Once it opens, paste:
|
||||
|
||||
```bash
|
||||
git clone --depth 1 https://github.com/BerriAI/litellm.git
|
||||
cd litellm/terraform/litellm/aws/examples/default && ./deploy.sh
|
||||
```
|
||||
|
||||
`deploy.sh` installs a pinned, checksum-verified Terraform (CloudShell doesn't
|
||||
ship one) and applies the stack. Full steps:
|
||||
[aws walkthrough](aws/examples/default/tutorial.md).
|
||||
|
||||
> **Trial vs. production.** The one-click roots serve **plain HTTP** and
|
||||
> register no models — fine for kicking the tires, not for production. For a
|
||||
> real deploy, add TLS (`acm_certificate_arn` on AWS, `lb_domains` on GCP),
|
||||
> register models via `proxy_config`, and supply your own master key. See the
|
||||
> per-stack READMEs.
|
||||
|
||||
## Components
|
||||
|
||||
The proxy is split into three deployables:
|
||||
|
||||
| Component | Default image | Port | Role |
|
||||
| --------- | ---------------------------------------- | ---- | -------------------------------------------------------------------- |
|
||||
| `gateway` | `ghcr.io/berriai/litellm-gateway:main-stable` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
|
||||
| `backend` | `ghcr.io/berriai/litellm-backend:main-stable` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
|
||||
| `ui` | `ghcr.io/berriai/litellm-ui:main-stable` | 3000 | Static Next.js dashboard served by nginx |
|
||||
| `gateway` | `ghcr.io/berriai/litellm-gateway:v1.86.0-dev` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
|
||||
| `backend` | `ghcr.io/berriai/litellm-backend:v1.86.0-dev` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
|
||||
| `ui` | `ghcr.io/berriai/litellm-ui:v1.86.0-dev` | 3000 | Static Next.js dashboard served by nginx |
|
||||
|
||||
The load balancer routes gateway path prefixes (mirrored verbatim from
|
||||
`gateway/routes/allowlist.py`) to the gateway, UI asset paths (`/`,
|
||||
|
|
@ -136,24 +181,30 @@ everything else to the backend.
|
|||
## Images
|
||||
|
||||
Both stacks take per-component image references as variables. The defaults
|
||||
point at the public `ghcr.io/berriai/litellm-<component>:main-stable`
|
||||
point at the public `ghcr.io/berriai/litellm-<component>:v1.86.0-dev`
|
||||
images, so the stack is runnable end-to-end without pre-flight setup —
|
||||
pin to a specific tag for production:
|
||||
|
||||
- **AWS** can pull from any registry the task execution role can reach.
|
||||
The role gets `AmazonECSTaskExecutionRolePolicy` attached, which grants
|
||||
ECR pull permissions for repositories in the same account.
|
||||
ECR pull permissions for repositories in the same account. GHCR is
|
||||
anonymous-readable, so the defaults work as-is.
|
||||
|
||||
- **GCP Cloud Run** can only pull from Artifact Registry or
|
||||
`gcr.io`-style registries. To use images hosted elsewhere, mirror them
|
||||
into Artifact Registry first.
|
||||
`gcr.io`-style registries — it rejects `ghcr.io`. By default the GCP
|
||||
stack auto-creates an Artifact Registry **remote repository** that
|
||||
proxies `https://ghcr.io` (`create_image_proxy_repo = true`), so the
|
||||
upstream images pull with no manual mirroring. Set
|
||||
`create_image_proxy_repo = false` and supply your own `image_registry`
|
||||
to opt out.
|
||||
|
||||
## Migrations
|
||||
|
||||
LiteLLM's proxy runs `prisma migrate deploy` at startup, but on first apply
|
||||
the gateway/backend can race the empty database. Both stacks expose a
|
||||
one-off migration task that runs `python litellm/proxy/prisma_migration.py`
|
||||
against the backend image:
|
||||
one-off migration task that runs `python3 /app/run.py` (assembles
|
||||
`DATABASE_URL` from the `DATABASE_*` env vars, then `prisma migrate deploy`)
|
||||
from the dedicated `ghcr.io/berriai/litellm-migrations` image:
|
||||
|
||||
- AWS: an `aws_ecs_task_definition` (`litellm-migrations`). Run with
|
||||
`aws ecs run-task` — the command is printed in `terraform output`.
|
||||
|
|
@ -165,11 +216,39 @@ gateway/backend services start serving traffic.
|
|||
|
||||
## What's not included
|
||||
|
||||
- TLS certificates / custom domains. Both stacks expose plain-HTTP load
|
||||
balancers; bring your own ACM cert (AWS) or managed cert (GCP) and wire
|
||||
it into the LB resource.
|
||||
- Custom domains / DNS. Both stacks support TLS out of the box — an ACM
|
||||
cert (`acm_certificate_arn`) on AWS, a Google-managed cert (`lb_domains`)
|
||||
on GCP — and `terraform plan` refuses to provision a plaintext LB unless
|
||||
you explicitly opt in (`allow_plaintext_alb` / `allow_plaintext_lb`,
|
||||
which the one-click trial roots default to true). You still bring your
|
||||
own DNS name and point it at the LB; see the per-stack "TLS" sections.
|
||||
- Remote state backends. Default local state — add an `s3` or `gcs`
|
||||
backend block to `versions.tf` when graduating to a team environment.
|
||||
- Observability beyond the cloud provider's defaults (CloudWatch logs on
|
||||
AWS, Cloud Logging on GCP). Wire your own Prometheus / Datadog / Langfuse
|
||||
via the `*_extra_env` variables.
|
||||
|
||||
## Publishing to the Terraform Registry
|
||||
|
||||
These modules are registry-conformant — each is self-contained, declares no
|
||||
`provider` block, ships a `README.md` + `examples/default/`, and documents
|
||||
every variable/output. The public registry only indexes a module at the
|
||||
**root** of a repo named `terraform-<PROVIDER>-<NAME>`, so the two stacks
|
||||
here are mirrored out to dedicated repos rather than published in place:
|
||||
|
||||
| Module | Mirror repo | Registry source |
|
||||
| ----------------------- | ----------------------------------- | ---------------------- |
|
||||
| `terraform/litellm/aws` | `BerriAI/terraform-aws-litellm` | `BerriAI/litellm/aws` |
|
||||
| `terraform/litellm/gcp` | `BerriAI/terraform-google-litellm` | `BerriAI/litellm/google` |
|
||||
|
||||
The [`Publish Terraform modules`](../../.github/workflows/terraform-modules-publish.yml)
|
||||
GitHub Actions workflow does the mirroring: it `git subtree split`s each
|
||||
module subdirectory into its mirror repo and tags it with the version you
|
||||
pass. Run it manually (Actions → Publish Terraform modules → enter `vX.Y.Z`)
|
||||
after a release. One-time setup (create the mirror repos, connect them to the
|
||||
registry, add the `TERRAFORM_REGISTRY_SYNC_TOKEN` secret) is documented in
|
||||
the workflow header.
|
||||
|
||||
Until a version is published to the registry, consume the modules straight
|
||||
from this repo with the `github.com/BerriAI/litellm//terraform/litellm/<stack>?ref=<tag>`
|
||||
source shown at the top.
|
||||
|
|
|
|||
|
|
@ -1,5 +1,22 @@
|
|||
# LiteLLM on AWS (ECS Fargate)
|
||||
|
||||
[](https://console.aws.amazon.com/cloudshell/home)
|
||||
|
||||
> **One-click trial:** open [AWS CloudShell](https://console.aws.amazon.com/cloudshell/home)
|
||||
> (your credentials are already there) and paste:
|
||||
>
|
||||
> ```bash
|
||||
> git clone --depth 1 https://github.com/BerriAI/litellm.git
|
||||
> cd litellm/terraform/litellm/aws/examples/default && ./deploy.sh
|
||||
> ```
|
||||
>
|
||||
> `deploy.sh` installs a pinned, checksum-verified Terraform and applies the
|
||||
> stack. A bare `terraform apply` from `examples/default/` also works with
|
||||
> **no tfvars** — it picks the first two AZs in `us-west-2`, serves plain
|
||||
> HTTP, and auto-generates a master key. Add `acm_certificate_arn` (TLS) and
|
||||
> `proxy_config` (models) for a real deployment. Full steps:
|
||||
> [examples/default/tutorial.md](examples/default/tutorial.md).
|
||||
|
||||
Deploys the componentized LiteLLM proxy on AWS:
|
||||
|
||||
- **VPC** with public + private subnets across the AZs you pass in, one NAT gateway
|
||||
|
|
@ -175,15 +192,24 @@ example files.
|
|||
|
||||
## Quick start
|
||||
|
||||
Zero-config trial (no tfvars needed — defaults to `us-west-2`, first two AZs,
|
||||
auto-generated master key, HTTP-only):
|
||||
|
||||
```bash
|
||||
cd terraform/litellm/aws/examples/default
|
||||
cp terraform.tfvars.example terraform.tfvars
|
||||
# Edit: region, tenant, env, azs, proxy_config, gateway_extra_secrets.
|
||||
|
||||
terraform init
|
||||
terraform apply
|
||||
```
|
||||
|
||||
To customize, drop in a tfvars file first:
|
||||
|
||||
```bash
|
||||
cp terraform.tfvars.example terraform.tfvars
|
||||
# Optional edits: region, tenant, env, azs, acm_certificate_arn (TLS),
|
||||
# proxy_config (models), gateway_extra_secrets (provider keys).
|
||||
terraform apply
|
||||
```
|
||||
|
||||
`examples/default/` is a thin root that configures the `aws` provider and
|
||||
calls the module (`../../`). It exposes a curated variable surface; for
|
||||
advanced knobs (per-component CPU/memory/workers, autoscaling, RDS/Redis
|
||||
|
|
|
|||
|
|
@ -38,6 +38,18 @@ resource "aws_lb_target_group" "gateway" {
|
|||
|
||||
deregistration_delay = 30
|
||||
|
||||
# AWS caps ELB / target-group names at 32 chars. local.name plus the
|
||||
# longest per-resource suffix ("-gateway", 8 chars) must fit, i.e.
|
||||
# length(tenant) + length(env) <= 15 (local.name = "<tenant>-litellm-<env>").
|
||||
# Checked here so a too-long tenant/env fails at plan with a clear message
|
||||
# instead of an opaque AWS API error deep into apply.
|
||||
lifecycle {
|
||||
precondition {
|
||||
condition = length(local.name) <= 24
|
||||
error_message = "Resource-name prefix '${local.name}' is ${length(local.name)} chars; with an 8-char suffix it exceeds the 32-char AWS load-balancer/target-group limit. Keep length(tenant) + length(env) <= 15."
|
||||
}
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -135,7 +135,13 @@ locals {
|
|||
command = [
|
||||
"python -c \"import os, base64, pathlib; pathlib.Path(os.environ['CONFIG_FILE_PATH']).write_bytes(base64.b64decode(os.environ['LITELLM_PROXY_CONFIG_B64']))\" && exec uvicorn backend.main:app ${local.backend_uvicorn_args}"
|
||||
]
|
||||
} : {}
|
||||
} : {
|
||||
# Pin the backend listener to 0.0.0.0:4001 even without a proxy_config,
|
||||
# so it always matches the target-group health check and container_port.
|
||||
# Mirrors the gateway's no-config branch above.
|
||||
entryPoint = ["uvicorn", "backend.main:app"]
|
||||
command = split(" ", local.backend_uvicorn_args)
|
||||
}
|
||||
}
|
||||
|
||||
# ---------- Gateway ----------
|
||||
|
|
|
|||
84
terraform/litellm/aws/examples/default/deploy.sh
Executable file
84
terraform/litellm/aws/examples/default/deploy.sh
Executable file
|
|
@ -0,0 +1,84 @@
|
|||
#!/usr/bin/env bash
|
||||
#
|
||||
# One-command LiteLLM deploy helper for AWS CloudShell (or any machine with
|
||||
# the AWS CLI installed and credentials configured).
|
||||
#
|
||||
# AWS CloudShell ships git + the AWS CLI but not Terraform, so this script
|
||||
# installs a pinned, checksum-verified Terraform into ./.bin if it isn't
|
||||
# already on PATH, then runs `terraform init` + `terraform apply` against the
|
||||
# trial root in this directory.
|
||||
#
|
||||
# Usage:
|
||||
# ./deploy.sh # interactive apply (review plan, type yes)
|
||||
# AUTO_APPROVE=1 ./deploy.sh # non-interactive
|
||||
#
|
||||
# Override the trial defaults with TF_VAR_* env vars or a terraform.tfvars
|
||||
# file, e.g.:
|
||||
# export TF_VAR_region=us-east-1
|
||||
# ./deploy.sh
|
||||
set -euo pipefail
|
||||
|
||||
TF_VERSION="1.9.8"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
cd "$SCRIPT_DIR"
|
||||
|
||||
# ---- Resolve a terraform binary (install a pinned one if missing) ----
|
||||
if command -v terraform >/dev/null 2>&1; then
|
||||
TF="terraform"
|
||||
else
|
||||
case "$(uname -m)" in
|
||||
x86_64 | amd64) ARCH="amd64" ;;
|
||||
aarch64 | arm64) ARCH="arm64" ;;
|
||||
*)
|
||||
echo "Unsupported architecture: $(uname -m)" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
BIN_DIR="$SCRIPT_DIR/.bin"
|
||||
TF="$BIN_DIR/terraform"
|
||||
if [ ! -x "$TF" ]; then
|
||||
echo "Installing Terraform ${TF_VERSION} (${ARCH}) into ${BIN_DIR} ..."
|
||||
mkdir -p "$BIN_DIR"
|
||||
tmp="$(mktemp -d)"
|
||||
base="https://releases.hashicorp.com/terraform/${TF_VERSION}"
|
||||
zip="terraform_${TF_VERSION}_linux_${ARCH}.zip"
|
||||
# Download the artifact and HashiCorp's official checksum sidecar, then
|
||||
# verify before unpacking — never trust an unverified download.
|
||||
curl -fsSL -o "$tmp/$zip" "$base/$zip"
|
||||
curl -fsSL -o "$tmp/SHA256SUMS" "$base/terraform_${TF_VERSION}_SHA256SUMS"
|
||||
(cd "$tmp" && grep " $zip\$" SHA256SUMS | sha256sum -c -)
|
||||
unzip -o "$tmp/$zip" -d "$BIN_DIR" >/dev/null
|
||||
rm -rf "$tmp"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "Using $("$TF" version | head -1)"
|
||||
|
||||
# ---- Sanity-check AWS credentials before spending 15+ minutes applying ----
|
||||
if ! aws sts get-caller-identity >/dev/null 2>&1; then
|
||||
echo "AWS credentials not found. In CloudShell this is automatic; otherwise run 'aws configure' or set AWS_PROFILE." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ---- Deploy ----
|
||||
"$TF" init -input=false
|
||||
if [ "${AUTO_APPROVE:-}" = "1" ]; then
|
||||
"$TF" apply -auto-approve -input=false
|
||||
else
|
||||
"$TF" apply -input=false
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "==================================================================="
|
||||
echo " LiteLLM is deploying. Useful outputs:"
|
||||
echo "==================================================================="
|
||||
URL="$("$TF" output -raw alb_url 2>/dev/null || true)"
|
||||
KEY_ARN="$("$TF" output -raw master_key_secret_arn 2>/dev/null || true)"
|
||||
echo " Proxy URL : ${URL:-<run: terraform output alb_url>}"
|
||||
echo " UI login : admin / <master key>"
|
||||
if [ -n "$KEY_ARN" ]; then
|
||||
echo " Master key: aws secretsmanager get-secret-value --secret-id $KEY_ARN --query SecretString --output text"
|
||||
fi
|
||||
echo
|
||||
echo "The ALB takes a few minutes to pass health checks after apply returns."
|
||||
|
|
@ -16,13 +16,24 @@
|
|||
#
|
||||
# Knobs not surfaced as variables here (per-component sizing, autoscaling,
|
||||
# RDS/Redis tuning) can be set directly on this block — see ../../variables.tf.
|
||||
|
||||
# When azs is left empty, pick the first two AZs in the region so a
|
||||
# zero-config apply works in any region without the caller naming them.
|
||||
data "aws_availability_zones" "available" {
|
||||
state = "available"
|
||||
}
|
||||
|
||||
locals {
|
||||
azs = length(var.azs) > 0 ? var.azs : slice(data.aws_availability_zones.available.names, 0, 2)
|
||||
}
|
||||
|
||||
module "litellm" {
|
||||
source = "../../"
|
||||
|
||||
region = var.region
|
||||
tenant = var.tenant
|
||||
env = var.env
|
||||
azs = var.azs
|
||||
azs = local.azs
|
||||
|
||||
litellm_master_key = var.litellm_master_key
|
||||
litellm_license = var.litellm_license
|
||||
|
|
|
|||
|
|
@ -1,12 +1,19 @@
|
|||
region = "us-west-2"
|
||||
azs = ["us-west-2a", "us-west-2b"]
|
||||
# EVERYTHING in this file is optional. With no tfvars at all, `terraform
|
||||
# apply` brings up a working HTTP-only trial instance: region defaults to
|
||||
# us-west-2, the first two AZs in the region are picked automatically,
|
||||
# tenant/env default to litellm/trial, and a master key is auto-generated.
|
||||
# Override any of the values below for a real deployment.
|
||||
|
||||
# region = "us-west-2"
|
||||
# azs = ["us-west-2a", "us-west-2b"] # empty → first two AZs in the region
|
||||
|
||||
# Resource naming: every AWS resource the stack creates is named
|
||||
# `${tenant}-litellm-${env}` (or that plus a per-resource suffix). E.g.
|
||||
# tenant="acme" + env="stage" → ALB `acme-litellm-stage`, ECS service
|
||||
# `acme-litellm-stage-gateway`, etc.
|
||||
tenant = "acme"
|
||||
env = "stage"
|
||||
# `acme-litellm-stage-gateway`, etc. Keep length(tenant)+length(env) <= 15
|
||||
# (AWS caps ELB/target-group names at 32 chars).
|
||||
# tenant = "acme"
|
||||
# env = "stage"
|
||||
|
||||
# Tenant-supplied secrets. Prefer TF_VAR_litellm_master_key /
|
||||
# TF_VAR_litellm_license / TF_VAR_ui_password env vars so the values don't
|
||||
|
|
@ -17,10 +24,10 @@ env = "stage"
|
|||
# litellm_license = "lic-..."
|
||||
# ui_password = "..."
|
||||
|
||||
# TLS: provide an ACM cert for production. Without one, plan fails unless
|
||||
# allow_plaintext_alb = true is set explicitly (trial/dev only).
|
||||
# TLS: this trial root defaults to HTTP-only (allow_plaintext_alb = true).
|
||||
# For a real deployment, provide an ACM cert and turn plaintext back off.
|
||||
# acm_certificate_arn = "arn:aws:acm:us-west-2:111122223333:certificate/..."
|
||||
# allow_plaintext_alb = true
|
||||
# allow_plaintext_alb = false
|
||||
|
||||
# Storage retention: false (default) makes `terraform destroy` refuse on a
|
||||
# non-empty bucket. Flip to true only for ephemeral / CI stacks.
|
||||
|
|
|
|||
80
terraform/litellm/aws/examples/default/tutorial.md
Normal file
80
terraform/litellm/aws/examples/default/tutorial.md
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
# Deploy LiteLLM on AWS
|
||||
|
||||
This walkthrough deploys the **LiteLLM AI Gateway** into your own AWS account
|
||||
using Terraform. You get the full componentized proxy — gateway, backend, and
|
||||
dashboard on ECS Fargate, fronted by an Application Load Balancer, backed by
|
||||
Aurora Postgres, ElastiCache (Redis), and an S3 bucket.
|
||||
|
||||
The fastest path is **AWS CloudShell**, which already has the AWS CLI and your
|
||||
credentials wired up. Click the button in the
|
||||
[module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/aws/README.md),
|
||||
or open <https://console.aws.amazon.com/cloudshell> and follow along.
|
||||
|
||||
## 1. Get the code
|
||||
|
||||
In CloudShell (or any machine with the AWS CLI configured):
|
||||
|
||||
```bash
|
||||
git clone --depth 1 https://github.com/BerriAI/litellm.git
|
||||
cd litellm/terraform/litellm/aws/examples/default
|
||||
```
|
||||
|
||||
## 2. Deploy
|
||||
|
||||
`deploy.sh` installs a pinned, checksum-verified Terraform (CloudShell doesn't
|
||||
ship one), then runs `terraform init` + `terraform apply`:
|
||||
|
||||
```bash
|
||||
./deploy.sh
|
||||
```
|
||||
|
||||
Review the plan and type `yes`. The apply provisions the VPC, Aurora cluster,
|
||||
Redis, S3, and the three ECS services, bootstraps the database, runs the
|
||||
schema migration, and only then starts the services — so it takes **15-20
|
||||
minutes** on the first run.
|
||||
|
||||
> Prefer to drive Terraform yourself? `terraform init && terraform apply` works
|
||||
> too, as long as Terraform is already installed.
|
||||
|
||||
## 3. You're live
|
||||
|
||||
```bash
|
||||
terraform output alb_url
|
||||
```
|
||||
|
||||
The dashboard is at `/`, the OpenAI-compatible API at `/v1/*`. Log in with
|
||||
username `admin` and the auto-generated master key:
|
||||
|
||||
```bash
|
||||
aws secretsmanager get-secret-value \
|
||||
--secret-id "$(terraform output -raw master_key_secret_arn)" \
|
||||
--query SecretString --output text
|
||||
```
|
||||
|
||||
The ALB takes a few minutes to pass health checks after apply returns.
|
||||
|
||||
## 4. Customize (optional)
|
||||
|
||||
This trial deploy serves plain HTTP and registers no models. For a real
|
||||
deployment, copy and edit the tfvars file, then re-apply:
|
||||
|
||||
```bash
|
||||
cp terraform.tfvars.example terraform.tfvars
|
||||
# Edit: region, tenant/env, acm_certificate_arn (TLS), proxy_config (models),
|
||||
# gateway_extra_secrets (provider API keys).
|
||||
terraform apply
|
||||
```
|
||||
|
||||
Provider API keys go in AWS Secrets Manager and are referenced by ARN — see
|
||||
the [module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/aws/README.md)
|
||||
for the full configuration surface (TLS, models, sizing, multi-tenant).
|
||||
|
||||
## 5. Clean up
|
||||
|
||||
```bash
|
||||
terraform destroy
|
||||
```
|
||||
|
||||
Aurora takes a final snapshot and the S3 bucket refuses to delete while
|
||||
non-empty (data-loss guards). Set `skip_final_snapshot = true` /
|
||||
`s3_force_destroy = true` for an ephemeral trial you don't mind losing.
|
||||
|
|
@ -5,24 +5,30 @@
|
|||
# per-variable docs live in ../../variables.tf — the module is the source
|
||||
# of truth; descriptions here are intentionally terse.
|
||||
|
||||
# Defaults make a bare `terraform apply` bring up a working trial instance
|
||||
# (no tfvars required). Override any of them for a real deployment.
|
||||
variable "region" {
|
||||
description = "AWS region to deploy into."
|
||||
type = string
|
||||
default = "us-west-2"
|
||||
}
|
||||
|
||||
variable "tenant" {
|
||||
description = "Tenant slug — prefix for every resource (<tenant>-litellm-<env>)."
|
||||
type = string
|
||||
default = "litellm"
|
||||
}
|
||||
|
||||
variable "env" {
|
||||
description = "Environment suffix (stage, prod, dev)."
|
||||
type = string
|
||||
default = "trial"
|
||||
}
|
||||
|
||||
variable "azs" {
|
||||
description = "Availability zones for subnets. At least 2 (RDS + ALB)."
|
||||
description = "Availability zones for subnets. At least 2 (RDS + ALB). Empty (default) auto-picks the first two AZs in the region."
|
||||
type = list(string)
|
||||
default = []
|
||||
}
|
||||
|
||||
# Sensitive — prefer TF_VAR_litellm_master_key / TF_VAR_litellm_license /
|
||||
|
|
@ -56,9 +62,9 @@ variable "acm_certificate_arn" {
|
|||
}
|
||||
|
||||
variable "allow_plaintext_alb" {
|
||||
description = "Opt into HTTP-only ALB (trial/dev only)."
|
||||
description = "Opt into HTTP-only ALB (trial/dev only). Defaults true in this trial root so a zero-config apply succeeds; set acm_certificate_arn (and flip this to false) for a real deployment."
|
||||
type = bool
|
||||
default = false
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "s3_force_destroy" {
|
||||
|
|
|
|||
|
|
@ -1,5 +1,14 @@
|
|||
# LiteLLM on GCP (Cloud Run)
|
||||
|
||||
[](https://shell.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https://github.com/BerriAI/litellm&cloudshell_workspace=terraform/litellm/gcp/examples/default&cloudshell_tutorial=tutorial.md)
|
||||
|
||||
> **One-click trial:** the button above opens Cloud Shell with a guided
|
||||
> walkthrough that deploys this stack into your active project. A bare
|
||||
> `terraform apply` from `examples/default/` also works with **no tfvars** —
|
||||
> it serves plain HTTP, auto-generates a master key, and auto-creates an
|
||||
> Artifact Registry proxy so Cloud Run can pull the images. Add `lb_domains`
|
||||
> (TLS) and `proxy_config` (models) for a real deployment.
|
||||
|
||||
Deploys the componentized LiteLLM proxy on GCP:
|
||||
|
||||
- **VPC** + Private Services Access range + a Serverless VPC Access connector
|
||||
|
|
@ -27,11 +36,21 @@ Bump them together when bumping LiteLLM.
|
|||
|
||||
Cloud Run only accepts images from Artifact Registry, `[region.]gcr.io`,
|
||||
or `docker.io` — `ghcr.io` URIs are rejected at apply time. The four
|
||||
images are published to GHCR upstream, so any real deploy needs an
|
||||
Artifact Registry remote repository pointed at GHCR.
|
||||
images are published to GHCR upstream, so a remote Artifact Registry
|
||||
repository is needed to proxy them.
|
||||
|
||||
**One-time setup (per project):** create a remote repo and let Cloud Run
|
||||
pull through it.
|
||||
**By default the stack creates this for you.** With
|
||||
`create_image_proxy_repo = true` (the default), Terraform provisions an
|
||||
Artifact Registry remote repository pointed at `https://ghcr.io`, grants
|
||||
the project's serverless agent read on it, and — when `image_registry` is
|
||||
left empty — composes the four image URIs from it automatically:
|
||||
`<region>-docker.pkg.dev/<project>/<tenant>-litellm-<env>-ghcr/berriai/litellm-<component>:<image_tag>`.
|
||||
That's what makes the zero-config deploy work end-to-end with no manual
|
||||
mirroring step.
|
||||
|
||||
**To use your own registry instead**, set `create_image_proxy_repo = false`
|
||||
and point `image_registry` at an existing Artifact Registry path. The
|
||||
manual equivalent of the auto-created repo is:
|
||||
|
||||
```bash
|
||||
gcloud artifacts repositories create litellm \
|
||||
|
|
@ -42,11 +61,10 @@ gcloud artifacts repositories create litellm \
|
|||
--remote-docker-repo=https://ghcr.io
|
||||
```
|
||||
|
||||
Then point the stack at it via `image_registry`:
|
||||
|
||||
```hcl
|
||||
image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
|
||||
image_tag = "v1.86.0-dev"
|
||||
create_image_proxy_repo = false
|
||||
image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
|
||||
image_tag = "v1.86.0-dev"
|
||||
```
|
||||
|
||||
The four `litellm-<component>:${image_tag}` URIs are composed from those
|
||||
|
|
@ -54,15 +72,8 @@ two vars. Set `gateway_image` / `backend_image` / `ui_image` /
|
|||
`migrations_image` only if you need a per-component override (custom
|
||||
build, different tag).
|
||||
|
||||
Two further notes:
|
||||
|
||||
- The runtime SAs the stack creates do **not** need
|
||||
`roles/artifactregistry.reader` — Cloud Run pulls images using the
|
||||
per-project serverless agent
|
||||
(`service-<project-num>@serverless-robot-prod.iam.gserviceaccount.com`),
|
||||
not the runtime SA.
|
||||
- For a fully air-gapped option, mirror the images into a regular AR
|
||||
repository instead of a remote repo:
|
||||
For a fully air-gapped option, mirror the images into a regular AR
|
||||
repository instead of a remote repo:
|
||||
|
||||
```bash
|
||||
for c in gateway backend ui migrations; do
|
||||
|
|
@ -204,15 +215,24 @@ example files.
|
|||
|
||||
## Quick start
|
||||
|
||||
Zero-config trial (no tfvars needed — project is inferred from your active
|
||||
gcloud/ADC project):
|
||||
|
||||
```bash
|
||||
cd terraform/litellm/gcp/examples/default
|
||||
cp terraform.tfvars.example terraform.tfvars
|
||||
# Edit: project, region, tenant, env, image_registry, proxy_config, gateway_extra_secrets.
|
||||
|
||||
terraform init
|
||||
terraform apply
|
||||
```
|
||||
|
||||
To customize, drop in a tfvars file first:
|
||||
|
||||
```bash
|
||||
cp terraform.tfvars.example terraform.tfvars
|
||||
# Optional edits: project, region, tenant, env, lb_domains (TLS),
|
||||
# proxy_config (models), gateway_extra_secrets (provider keys).
|
||||
terraform apply
|
||||
```
|
||||
|
||||
`examples/default/` is a thin root that configures the `google` /
|
||||
`google-beta` providers and calls the module (`../../`). It exposes a
|
||||
curated variable surface; for advanced knobs (per-component
|
||||
|
|
|
|||
49
terraform/litellm/gcp/artifact_registry.tf
Normal file
49
terraform/litellm/gcp/artifact_registry.tf
Normal file
|
|
@ -0,0 +1,49 @@
|
|||
# Cloud Run can only pull from Artifact Registry, [region.]gcr.io, or
|
||||
# docker.io — it rejects ghcr.io URIs at apply time. The four LiteLLM
|
||||
# images live on GHCR upstream, so by default this stack provisions an
|
||||
# Artifact Registry *remote repository* that transparently proxies
|
||||
# https://ghcr.io. Cloud Run then pulls `…-docker.pkg.dev/<project>/<repo>/
|
||||
# berriai/litellm-<component>` and AR fetches+caches from GHCR on first pull.
|
||||
#
|
||||
# This is what makes a zero-config deploy possible: with the proxy in place
|
||||
# the default `image_registry` resolves to the proxy path (see locals.tf),
|
||||
# so no manual `gcloud artifacts repositories create` step is needed.
|
||||
#
|
||||
# Set create_image_proxy_repo = false (and supply your own image_registry /
|
||||
# *_image) to skip it — e.g. when mirroring images into a standard AR repo.
|
||||
|
||||
data "google_project" "this" {
|
||||
project_id = var.project
|
||||
}
|
||||
|
||||
resource "google_artifact_registry_repository" "ghcr_proxy" {
|
||||
count = var.create_image_proxy_repo ? 1 : 0
|
||||
|
||||
location = var.region
|
||||
repository_id = "${local.name}-ghcr"
|
||||
description = "GitHub Container Registry (ghcr.io) passthrough for LiteLLM images"
|
||||
format = "DOCKER"
|
||||
mode = "REMOTE_REPOSITORY"
|
||||
labels = var.labels
|
||||
|
||||
remote_repository_config {
|
||||
description = "ghcr.io"
|
||||
docker_repository {
|
||||
custom_repository {
|
||||
uri = "https://ghcr.io"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# Cloud Run pulls images with the per-project serverless service agent, not
|
||||
# the runtime SA. Grant that agent read on the proxy repo so the pull (and
|
||||
# the upstream fetch) succeeds.
|
||||
resource "google_artifact_registry_repository_iam_member" "serverless_agent_reader" {
|
||||
count = var.create_image_proxy_repo ? 1 : 0
|
||||
|
||||
location = google_artifact_registry_repository.ghcr_proxy[0].location
|
||||
repository = google_artifact_registry_repository.ghcr_proxy[0].name
|
||||
role = "roles/artifactregistry.reader"
|
||||
member = "serviceAccount:service-${data.google_project.this.number}@serverless-robot-prod.iam.gserviceaccount.com"
|
||||
}
|
||||
|
|
@ -117,6 +117,16 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
name = "${local.name}-gateway"
|
||||
location = var.region
|
||||
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
|
||||
labels = var.labels
|
||||
|
||||
# Cloud Run rejects ghcr.io images. Catch the one misconfiguration that
|
||||
# otherwise fails deep into apply: registry cleared AND proxy disabled.
|
||||
lifecycle {
|
||||
precondition {
|
||||
condition = !startswith(local.gateway_image, "ghcr.io/") && !startswith(local.backend_image, "ghcr.io/") && !startswith(local.ui_image, "ghcr.io/") && !startswith(local.migrations_image, "ghcr.io/")
|
||||
error_message = "Cloud Run cannot pull from ghcr.io. Keep create_image_proxy_repo = true (default) to auto-create an Artifact Registry remote repo, or set image_registry to an Artifact Registry path."
|
||||
}
|
||||
}
|
||||
|
||||
template {
|
||||
service_account = google_service_account.runtime.email
|
||||
|
|
@ -197,6 +207,9 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
google_secret_manager_secret_iam_member.license,
|
||||
google_secret_manager_secret_iam_member.extras,
|
||||
google_sql_user.app,
|
||||
# The serverless agent needs read on the image-proxy repo before the
|
||||
# first pull (no-op when create_image_proxy_repo = false).
|
||||
google_artifact_registry_repository_iam_member.serverless_agent_reader,
|
||||
# Don't go live until the schema is migrated; otherwise the proxy boots,
|
||||
# fails on missing tables, and Cloud Run keeps cold-restarting.
|
||||
terraform_data.migration,
|
||||
|
|
@ -208,6 +221,7 @@ resource "google_cloud_run_v2_service" "backend" {
|
|||
name = "${local.name}-backend"
|
||||
location = var.region
|
||||
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
|
||||
labels = var.labels
|
||||
|
||||
template {
|
||||
service_account = google_service_account.runtime.email
|
||||
|
|
@ -289,6 +303,7 @@ resource "google_cloud_run_v2_service" "backend" {
|
|||
google_secret_manager_secret_iam_member.ui_password,
|
||||
google_secret_manager_secret_iam_member.extras,
|
||||
google_sql_user.app,
|
||||
google_artifact_registry_repository_iam_member.serverless_agent_reader,
|
||||
terraform_data.migration,
|
||||
]
|
||||
}
|
||||
|
|
@ -301,6 +316,7 @@ resource "google_cloud_run_v2_service" "ui" {
|
|||
name = "${local.name}-ui"
|
||||
location = var.region
|
||||
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
|
||||
labels = var.labels
|
||||
|
||||
template {
|
||||
service_account = google_service_account.ui_runtime.email
|
||||
|
|
@ -337,6 +353,10 @@ resource "google_cloud_run_v2_service" "ui" {
|
|||
}
|
||||
}
|
||||
}
|
||||
|
||||
depends_on = [
|
||||
google_artifact_registry_repository_iam_member.serverless_agent_reader,
|
||||
]
|
||||
}
|
||||
|
||||
# Allow the LB (any unauthenticated traffic from the configured serverless
|
||||
|
|
@ -374,6 +394,7 @@ resource "google_cloud_run_v2_service_iam_member" "ui_allusers" {
|
|||
resource "google_cloud_run_v2_job" "migrations" {
|
||||
name = "${local.name}-migrations"
|
||||
location = var.region
|
||||
labels = var.labels
|
||||
|
||||
template {
|
||||
template {
|
||||
|
|
@ -426,5 +447,6 @@ resource "google_cloud_run_v2_job" "migrations" {
|
|||
depends_on = [
|
||||
google_secret_manager_secret_iam_member.db_password,
|
||||
google_sql_user.app,
|
||||
google_artifact_registry_repository_iam_member.serverless_agent_reader,
|
||||
]
|
||||
}
|
||||
|
|
|
|||
|
|
@ -25,6 +25,7 @@ resource "google_sql_database_instance" "writer" {
|
|||
availability_type = "REGIONAL"
|
||||
disk_size = 20
|
||||
disk_autoresize = true
|
||||
user_labels = var.labels
|
||||
|
||||
backup_configuration {
|
||||
enabled = true
|
||||
|
|
@ -69,6 +70,7 @@ resource "google_sql_database_instance" "reader" {
|
|||
tier = var.db_tier
|
||||
availability_type = "ZONAL"
|
||||
disk_autoresize = true
|
||||
user_labels = var.labels
|
||||
|
||||
ip_configuration {
|
||||
ipv4_enabled = false
|
||||
|
|
|
|||
|
|
@ -17,10 +17,20 @@
|
|||
# Knobs not surfaced as variables here (per-component sizing/instances,
|
||||
# Cloud SQL tier/edition, Memorystore tier, per-component image overrides)
|
||||
# can be set directly on this block — see ../../variables.tf.
|
||||
|
||||
# Resolve the effective project: explicit var.project wins, otherwise fall
|
||||
# back to whatever the provider inferred from gcloud / ADC (Cloud Shell sets
|
||||
# this). The module needs a concrete project ID for project-scoped IAM.
|
||||
data "google_client_config" "current" {}
|
||||
|
||||
locals {
|
||||
project = var.project != "" ? var.project : data.google_client_config.current.project
|
||||
}
|
||||
|
||||
module "litellm" {
|
||||
source = "../../"
|
||||
|
||||
project = var.project
|
||||
project = local.project
|
||||
region = var.region
|
||||
tenant = var.tenant
|
||||
env = var.env
|
||||
|
|
|
|||
|
|
@ -6,12 +6,16 @@
|
|||
# The module's resources inherit these default (unaliased) `google` /
|
||||
# `google-beta` configs automatically through the module call, so project
|
||||
# and region set here flow into every resource that doesn't pass its own.
|
||||
# project = null when var.project is empty, which lets the provider infer
|
||||
# the project from the active gcloud config / ADC (set in Cloud Shell). The
|
||||
# resolved value is read back via data.google_client_config in main.tf and
|
||||
# passed explicitly to the module.
|
||||
provider "google" {
|
||||
project = var.project
|
||||
project = var.project != "" ? var.project : null
|
||||
region = var.region
|
||||
}
|
||||
|
||||
provider "google-beta" {
|
||||
project = var.project
|
||||
project = var.project != "" ? var.project : null
|
||||
region = var.region
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,12 +1,19 @@
|
|||
project = "my-gcp-project"
|
||||
region = "us-central1"
|
||||
# EVERYTHING in this file is optional. With no tfvars at all, `terraform
|
||||
# apply` brings up a working HTTP-only trial instance: project is inferred
|
||||
# from your active gcloud/ADC project, region defaults to us-central1,
|
||||
# tenant/env default to litellm/trial, and the module auto-creates an
|
||||
# Artifact Registry proxy so Cloud Run can pull the ghcr.io images. Override
|
||||
# any of the values below for a real deployment.
|
||||
|
||||
# project = "my-gcp-project" # empty → inferred from gcloud / ADC
|
||||
# region = "us-central1"
|
||||
|
||||
# Resource naming: every GCP resource the stack creates is named
|
||||
# `${tenant}-litellm-${env}` (or that plus a per-resource suffix). E.g.
|
||||
# tenant="acme" + env="stage" → Cloud Run service `acme-litellm-stage-gateway`,
|
||||
# Cloud SQL instance `acme-litellm-stage`, etc.
|
||||
tenant = "acme"
|
||||
env = "stage"
|
||||
# tenant = "acme"
|
||||
# env = "stage"
|
||||
|
||||
# Tenant-supplied secrets. Prefer TF_VAR_litellm_master_key /
|
||||
# TF_VAR_litellm_license / TF_VAR_ui_password env vars so the values don't
|
||||
|
|
@ -17,23 +24,23 @@ env = "stage"
|
|||
# litellm_license = "lic-..."
|
||||
# ui_password = "..."
|
||||
|
||||
# TLS: provide DNS names already pointing at the LB IP for a Google-managed
|
||||
# cert. Without one, plan fails unless allow_plaintext_lb = true is set
|
||||
# explicitly (trial/dev only).
|
||||
# TLS: this trial root defaults to HTTP-only (allow_plaintext_lb = true).
|
||||
# For a real deployment, provide DNS names already pointing at the LB IP for
|
||||
# a Google-managed cert and turn plaintext back off.
|
||||
# lb_domains = ["proxy.example.com"]
|
||||
# allow_plaintext_lb = true
|
||||
# allow_plaintext_lb = false
|
||||
|
||||
# Storage and database retention. Defaults are safe — destroy preserves
|
||||
# data. Flip these only for ephemeral / CI stacks.
|
||||
# cloudsql_deletion_protection = true # default: refuse destroy on the DB
|
||||
# gcs_force_destroy = false # default: refuse destroy on a non-empty bucket
|
||||
|
||||
# Images. Cloud Run rejects ghcr.io, so a real deploy must point
|
||||
# image_registry at an Artifact Registry remote repo (see README "Image
|
||||
# pulls"); image_tag is applied to all four litellm-* images. Per-component
|
||||
# *_image overrides are NOT exposed here — set them directly on the
|
||||
# `module "litellm"` block in main.tf (see ../../variables.tf) if you need
|
||||
# to mix-and-match versions.
|
||||
# Images. Left empty, the module auto-creates an Artifact Registry remote
|
||||
# repo proxying ghcr.io (Cloud Run rejects ghcr.io directly), so the default
|
||||
# images pull with no setup. Point image_registry at your own Artifact
|
||||
# Registry repo to bypass the proxy; image_tag applies to all four litellm-*
|
||||
# images. Per-component *_image overrides are NOT exposed here — set them on
|
||||
# the `module "litellm"` block in main.tf (see ../../variables.tf).
|
||||
# image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
|
||||
# image_tag = "v1.86.0-dev"
|
||||
|
||||
|
|
|
|||
131
terraform/litellm/gcp/examples/default/tutorial.md
Normal file
131
terraform/litellm/gcp/examples/default/tutorial.md
Normal file
|
|
@ -0,0 +1,131 @@
|
|||
# Deploy LiteLLM on Google Cloud
|
||||
|
||||
<walkthrough-tutorial-duration duration="20"></walkthrough-tutorial-duration>
|
||||
|
||||
This guided walkthrough deploys the **LiteLLM AI Gateway** into your own
|
||||
Google Cloud project using Terraform. You get the full componentized proxy —
|
||||
gateway, backend, and dashboard on Cloud Run, fronted by an external HTTP(S)
|
||||
load balancer, backed by Cloud SQL (Postgres), Memorystore (Redis), and a
|
||||
GCS bucket.
|
||||
|
||||
Everything runs from this Cloud Shell session — Terraform is already
|
||||
installed here, so there's nothing to set up on your machine.
|
||||
|
||||
## Choose your project
|
||||
|
||||
Pick the Google Cloud project to deploy into. Everything the stack creates is
|
||||
billed to and lives in this project.
|
||||
|
||||
<walkthrough-project-setup></walkthrough-project-setup>
|
||||
|
||||
Set it as the active project for this session:
|
||||
|
||||
```bash
|
||||
gcloud config set project <walkthrough-project-id/>
|
||||
```
|
||||
|
||||
## Enable the required APIs
|
||||
|
||||
The stack uses Cloud Run, Cloud SQL, Memorystore, Secret Manager, Serverless
|
||||
VPC Access, Compute, Service Networking, Cloud Storage, and Artifact Registry.
|
||||
Enable them all in one call (this can take a minute):
|
||||
|
||||
```bash
|
||||
gcloud services enable \
|
||||
run.googleapis.com \
|
||||
sqladmin.googleapis.com \
|
||||
redis.googleapis.com \
|
||||
secretmanager.googleapis.com \
|
||||
vpcaccess.googleapis.com \
|
||||
compute.googleapis.com \
|
||||
servicenetworking.googleapis.com \
|
||||
storage.googleapis.com \
|
||||
artifactregistry.googleapis.com
|
||||
```
|
||||
|
||||
## Deploy
|
||||
|
||||
Tell Terraform which project to use, then initialize and apply. The apply
|
||||
provisions everything, runs the database migration, and only then starts the
|
||||
services — so it takes **15-20 minutes** on the first run (Cloud SQL and the
|
||||
load balancer are the slow parts).
|
||||
|
||||
```bash
|
||||
export TF_VAR_project=$(gcloud config get-value project)
|
||||
terraform init
|
||||
terraform apply
|
||||
```
|
||||
|
||||
Review the plan and type `yes` to proceed.
|
||||
|
||||
<walkthrough-footnote>This trial deploy serves plain HTTP and auto-generates
|
||||
a master key. For a production deploy, set `lb_domains` for a managed TLS
|
||||
cert and supply your own master key — see the module README.</walkthrough-footnote>
|
||||
|
||||
## You're live
|
||||
|
||||
Print the proxy URL (the dashboard is at `/`, the OpenAI-compatible API at
|
||||
`/v1/*`):
|
||||
|
||||
```bash
|
||||
terraform output lb_url
|
||||
```
|
||||
|
||||
Fetch the auto-generated admin / master key — log into the dashboard with
|
||||
username `admin` and this value:
|
||||
|
||||
```bash
|
||||
gcloud secrets versions access latest \
|
||||
--secret="$(terraform output -raw master_key_secret_id)"
|
||||
```
|
||||
|
||||
Send a test request (replace `URL` and `KEY` with the two values above):
|
||||
|
||||
```bash
|
||||
curl "$(terraform output -raw lb_url)/v1/models" \
|
||||
-H "Authorization: Bearer $(gcloud secrets versions access latest --secret="$(terraform output -raw master_key_secret_id)")"
|
||||
```
|
||||
|
||||
## Add a model
|
||||
|
||||
Edit `terraform.tfvars` to register models and provider keys, then re-apply.
|
||||
Store provider API keys in Secret Manager and reference them, e.g.:
|
||||
|
||||
```bash
|
||||
echo -n "sk-proj-..." | gcloud secrets create openai-api-key --data-file=-
|
||||
```
|
||||
|
||||
```hcl
|
||||
proxy_config = {
|
||||
model_list = [{
|
||||
model_name = "gpt-4o"
|
||||
litellm_params = { model = "openai/gpt-4o", api_key = "os.environ/OPENAI_API_KEY" }
|
||||
}]
|
||||
}
|
||||
gateway_extra_secrets = {
|
||||
OPENAI_API_KEY = "projects/<walkthrough-project-id/>/secrets/openai-api-key"
|
||||
}
|
||||
```
|
||||
|
||||
Then `terraform apply` again.
|
||||
|
||||
## Clean up
|
||||
|
||||
To tear everything down when you're done:
|
||||
|
||||
```bash
|
||||
terraform destroy
|
||||
```
|
||||
|
||||
<walkthrough-footnote>Cloud SQL has deletion protection on by default, so
|
||||
destroy will refuse until you set `cloudsql_deletion_protection = false` (and
|
||||
`gcs_force_destroy = true` for a non-empty bucket) and re-apply. That's a
|
||||
guard against accidental data loss.</walkthrough-footnote>
|
||||
|
||||
## Done
|
||||
|
||||
<walkthrough-conclusion-trophy></walkthrough-conclusion-trophy>
|
||||
|
||||
You've deployed LiteLLM on Google Cloud. For configuration options (models,
|
||||
TLS, sizing, multi-tenant deploys), see the
|
||||
[module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/gcp/README.md).
|
||||
|
|
@ -5,9 +5,13 @@
|
|||
# main.tf, or call the module from your own root config. Full per-variable
|
||||
# docs live in ../../variables.tf — the module is the source of truth.
|
||||
|
||||
# Defaults make a bare `terraform apply` bring up a working trial instance.
|
||||
# `project` is the one value that has no safe default — empty means "infer
|
||||
# from the active gcloud/ADC project" (which Cloud Shell sets for you).
|
||||
variable "project" {
|
||||
description = "GCP project ID."
|
||||
description = "GCP project ID. Empty (default) infers the active gcloud / ADC project (set automatically in Cloud Shell)."
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "region" {
|
||||
|
|
@ -19,11 +23,13 @@ variable "region" {
|
|||
variable "tenant" {
|
||||
description = "Tenant slug — prefix for every resource (<tenant>-litellm-<env>)."
|
||||
type = string
|
||||
default = "litellm"
|
||||
}
|
||||
|
||||
variable "env" {
|
||||
description = "Environment suffix (stage, prod, dev)."
|
||||
type = string
|
||||
default = "trial"
|
||||
}
|
||||
|
||||
# Sensitive — prefer TF_VAR_litellm_master_key / TF_VAR_litellm_license /
|
||||
|
|
@ -49,13 +55,14 @@ variable "ui_password" {
|
|||
sensitive = true
|
||||
}
|
||||
|
||||
# Image source. Cloud Run rejects ghcr.io, so a real deploy must point
|
||||
# image_registry at an Artifact Registry remote repo (see README "Image
|
||||
# pulls"). Per-component overrides live in ../../variables.tf.
|
||||
# Image source. Empty (default) makes the module auto-create an Artifact
|
||||
# Registry remote repo proxying ghcr.io (Cloud Run rejects ghcr.io directly),
|
||||
# so images pull with no manual setup. Set this to your own Artifact Registry
|
||||
# path to bypass the proxy. Per-component overrides live in ../../variables.tf.
|
||||
variable "image_registry" {
|
||||
description = "Registry path prefix; images composed as <image_registry>/litellm-<component>:<image_tag>."
|
||||
description = "Registry path prefix; images composed as <image_registry>/litellm-<component>:<image_tag>. Empty → auto ghcr.io proxy repo."
|
||||
type = string
|
||||
default = "ghcr.io/berriai"
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "image_tag" {
|
||||
|
|
@ -72,9 +79,9 @@ variable "lb_domains" {
|
|||
}
|
||||
|
||||
variable "allow_plaintext_lb" {
|
||||
description = "Opt into HTTP-only LB (trial/dev only)."
|
||||
description = "Opt into HTTP-only LB (trial/dev only). Defaults true in this trial root so a zero-config apply succeeds; set lb_domains (and flip this to false) for a real deployment."
|
||||
type = bool
|
||||
default = false
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "cloudsql_deletion_protection" {
|
||||
|
|
|
|||
|
|
@ -69,11 +69,22 @@ locals {
|
|||
{ name = "CONFIG_FILE_PATH", value = "/tmp/litellm-config.yaml" },
|
||||
] : []
|
||||
|
||||
# Effective registry prefix. An explicit image_registry wins; otherwise,
|
||||
# when the ghcr.io proxy repo is created, images resolve to it (mirrors the
|
||||
# upstream `berriai/litellm-*` path under the remote repo). Falling all the
|
||||
# way through to ghcr.io is only reachable when the operator both clears
|
||||
# image_registry and disables the proxy — Cloud Run rejects it, so the
|
||||
# per-service precondition (cloudrun.tf) fails fast with guidance.
|
||||
image_registry = (
|
||||
var.image_registry != "" ? var.image_registry :
|
||||
var.create_image_proxy_repo ? "${var.region}-docker.pkg.dev/${var.project}/${google_artifact_registry_repository.ghcr_proxy[0].repository_id}/berriai" :
|
||||
"ghcr.io/berriai"
|
||||
)
|
||||
|
||||
# Resolved image URIs: per-component override wins, otherwise compose
|
||||
# from image_registry + image_tag. Cloud Run only accepts AR / gcr.io /
|
||||
# docker.io paths — see variables.tf for the full constraint list.
|
||||
gateway_image = var.gateway_image != "" ? var.gateway_image : "${var.image_registry}/litellm-gateway:${var.image_tag}"
|
||||
backend_image = var.backend_image != "" ? var.backend_image : "${var.image_registry}/litellm-backend:${var.image_tag}"
|
||||
ui_image = var.ui_image != "" ? var.ui_image : "${var.image_registry}/litellm-ui:${var.image_tag}"
|
||||
migrations_image = var.migrations_image != "" ? var.migrations_image : "${var.image_registry}/litellm-migrations:${var.image_tag}"
|
||||
# from the effective registry + image_tag.
|
||||
gateway_image = var.gateway_image != "" ? var.gateway_image : "${local.image_registry}/litellm-gateway:${var.image_tag}"
|
||||
backend_image = var.backend_image != "" ? var.backend_image : "${local.image_registry}/litellm-backend:${var.image_tag}"
|
||||
ui_image = var.ui_image != "" ? var.ui_image : "${local.image_registry}/litellm-ui:${var.image_tag}"
|
||||
migrations_image = var.migrations_image != "" ? var.migrations_image : "${local.image_registry}/litellm-migrations:${var.image_tag}"
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ resource "google_redis_instance" "this" {
|
|||
tier = var.redis_tier
|
||||
memory_size_gb = var.redis_memory_size_gb
|
||||
region = var.region
|
||||
labels = var.labels
|
||||
|
||||
authorized_network = google_compute_network.this.id
|
||||
connect_mode = "PRIVATE_SERVICE_ACCESS"
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ variable "env" {
|
|||
}
|
||||
|
||||
variable "labels" {
|
||||
description = "Resource labels merged into every label-supporting resource."
|
||||
description = "Resource labels applied to the billable / filterable resources: the three Cloud Run services, the migrations job, Cloud SQL (writer + reader), Memorystore, the GCS bucket, and the image-proxy Artifact Registry repo. (Compute networking and Secret Manager resources don't carry labels.)"
|
||||
type = map(string)
|
||||
default = {
|
||||
"managed-by" = "terraform"
|
||||
|
|
@ -104,19 +104,35 @@ variable "vpc_connector_cidr" {
|
|||
# pointed at ghcr.io (e.g. `us-central1-docker.pkg.dev/my-proj/litellm/berriai`)
|
||||
# or override the per-component `*_image` vars individually with full URIs.
|
||||
|
||||
variable "create_image_proxy_repo" {
|
||||
description = <<-EOT
|
||||
Create an Artifact Registry remote repository that proxies
|
||||
`https://ghcr.io`, so Cloud Run (which rejects ghcr.io URIs) can pull
|
||||
the upstream LiteLLM images without a manual mirroring step. Default
|
||||
true — this is what lets a zero-config deploy run end-to-end. When true
|
||||
and `image_registry` is left empty, the four image URIs resolve to the
|
||||
proxy repo automatically (see locals.tf). Set false if you supply your
|
||||
own `image_registry` / `*_image` (e.g. a standard AR repo you mirror
|
||||
into). Requires the `artifactregistry.googleapis.com` API.
|
||||
EOT
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "image_registry" {
|
||||
description = <<-EOT
|
||||
Registry path prefix used to compose the four LiteLLM image URIs as
|
||||
`<image_registry>/litellm-<component>:<image_tag>`. The default
|
||||
(`ghcr.io/berriai`) only works on registries Cloud Run accepts — for
|
||||
GHCR-backed deploys, create an Artifact Registry remote repository
|
||||
pointed at `https://ghcr.io` and set this to that repo's path
|
||||
(e.g. `us-central1-docker.pkg.dev/<project>/<remote-repo>/berriai`).
|
||||
`<image_registry>/litellm-<component>:<image_tag>`. Empty (default)
|
||||
resolves to the auto-created ghcr.io proxy repo when
|
||||
`create_image_proxy_repo = true`; otherwise it falls back to
|
||||
`ghcr.io/berriai` (which Cloud Run rejects — a precondition catches
|
||||
this). For a custom registry, set this to an Artifact Registry path
|
||||
(e.g. `us-central1-docker.pkg.dev/<project>/<repo>/berriai`).
|
||||
Per-component overrides (`gateway_image`, `backend_image`, `ui_image`,
|
||||
`migrations_image`) bypass this entirely when set.
|
||||
EOT
|
||||
type = string
|
||||
default = "ghcr.io/berriai"
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "image_tag" {
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue