diff --git a/.github/workflows/terraform-modules-publish.yml b/.github/workflows/terraform-modules-publish.yml
new file mode 100644
index 00000000000..75cb1251d2b
--- /dev/null
+++ b/.github/workflows/terraform-modules-publish.yml
@@ -0,0 +1,98 @@
+# Publishes the LiteLLM Terraform modules to the public Terraform Registry.
+#
+# The registry only indexes a module that lives at the ROOT of a repo named
+# `terraform--`, so the two modules in this monorepo
+# (terraform/litellm/aws, terraform/litellm/gcp) can't be published in place.
+# This workflow mirrors each module subtree out to its own dedicated repo and
+# tags it, so the registry picks up the new version.
+#
+# ── One-time setup ───────────────────────────────────────────────────────────
+# 1. Create two empty GitHub repos:
+# BerriAI/terraform-aws-litellm
+# BerriAI/terraform-google-litellm
+# 2. Connect each to the Terraform Registry (registry.terraform.io → Publish →
+# Module) once; subsequent tags are ingested automatically.
+# 3. Add a repo secret `TERRAFORM_REGISTRY_SYNC_TOKEN` — a PAT (or fine-grained
+# token) with `contents:write` on both mirror repos. Without it this
+# workflow no-ops.
+#
+# ── Usage ────────────────────────────────────────────────────────────────────
+# Actions → "Publish Terraform modules" → Run workflow, enter the version
+# (e.g. v1.86.0). Leave dry_run checked first to see what it would push.
+name: Publish Terraform modules
+
+on:
+ workflow_dispatch:
+ inputs:
+ version:
+ description: "Module version tag to publish (semver, e.g. v1.86.0)"
+ required: true
+ type: string
+ dry_run:
+ description: "Print actions without pushing"
+ required: false
+ type: boolean
+ default: true
+
+permissions:
+ contents: read
+
+jobs:
+ publish:
+ runs-on: ubuntu-latest
+ strategy:
+ fail-fast: false
+ matrix:
+ include:
+ - prefix: terraform/litellm/aws
+ repo: terraform-aws-litellm
+ - prefix: terraform/litellm/gcp
+ repo: terraform-google-litellm
+ steps:
+ - name: Validate version input
+ run: |
+ set -euo pipefail
+ if ! printf '%s' "${{ inputs.version }}" | grep -Eq '^v[0-9]+\.[0-9]+\.[0-9]+(-[0-9A-Za-z.-]+)?$'; then
+ echo "version must be semver with a leading v (e.g. v1.86.0)" >&2
+ exit 1
+ fi
+
+ - name: Checkout (full history for subtree split)
+ uses: actions/checkout@v4
+ with:
+ fetch-depth: 0
+
+ - name: Mirror ${{ matrix.prefix }} → BerriAI/${{ matrix.repo }}
+ env:
+ TOKEN: ${{ secrets.TERRAFORM_REGISTRY_SYNC_TOKEN }}
+ PREFIX: ${{ matrix.prefix }}
+ REPO: ${{ matrix.repo }}
+ VERSION: ${{ inputs.version }}
+ DRY_RUN: ${{ inputs.dry_run }}
+ run: |
+ set -euo pipefail
+ if [ -z "${TOKEN:-}" ]; then
+ echo "TERRAFORM_REGISTRY_SYNC_TOKEN not set — skipping (see workflow header for setup)."
+ exit 0
+ fi
+
+ git config user.name "litellm-release-bot"
+ git config user.email "release-bot@litellm.ai"
+
+ # Split the module subdirectory into a standalone commit graph whose
+ # root is the module itself (what the registry expects).
+ split_sha="$(git subtree split --prefix="$PREFIX" HEAD)"
+ echo "subtree split for $PREFIX -> $split_sha"
+
+ remote="https://x-access-token:${TOKEN}@github.com/BerriAI/${REPO}.git"
+
+ if [ "$DRY_RUN" = "true" ]; then
+ echo "[dry-run] would push $split_sha to ${REPO}:main and tag $VERSION"
+ exit 0
+ fi
+
+ # Publish the split as the mirror's main branch + the version tag.
+ git push --force "$remote" "${split_sha}:refs/heads/main"
+ git tag -f "$VERSION" "$split_sha"
+ git push --force "$remote" "refs/tags/${VERSION}"
+ echo "Published BerriAI/${REPO}@${VERSION}"
diff --git a/README.md b/README.md
index 72fd43925c9..5eb90c1d4c2 100644
--- a/README.md
+++ b/README.md
@@ -10,7 +10,11 @@
+
+
+ One-click cloud deploys (Terraform) →
+ terraform/litellm
diff --git a/terraform/litellm/README.md b/terraform/litellm/README.md
index f1fa455f65e..ecb7e368f14 100644
--- a/terraform/litellm/README.md
+++ b/terraform/litellm/README.md
@@ -13,10 +13,18 @@ one-command deploy path. To embed a stack in your own config, call the module
by source:
```hcl
+# Direct from this monorepo (works today, no registry needed):
module "litellm" {
source = "github.com/BerriAI/litellm//terraform/litellm/aws?ref="
# ... inputs ...
}
+
+# Or, once published (see "Publishing to the Terraform Registry" below):
+module "litellm" {
+ source = "BerriAI/litellm/aws" # registry shorthand
+ version = "~> 1.86"
+ # ... inputs ...
+}
```
| Stack | Compute | Database (writer + reader) | Cache | Object store | Public entrypoint |
@@ -30,15 +38,52 @@ Both stacks support a typed `proxy_config` input (mirrors `helm/litellm`'s
`gateway.config.proxy_config`) and per-component extra env vars /
secret-manager refs.
+## One-click deploy
+
+The `examples/default/` roots are **zero-config trial deploys**: a bare
+`terraform apply` with no tfvars brings up a working (HTTP-only) instance —
+sensible region/zone defaults, an auto-generated master key, and (on GCP) an
+auto-created Artifact Registry proxy so Cloud Run can pull the images. Launch
+straight from your browser:
+
+### GCP — Open in Cloud Shell
+
+[](https://shell.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https://github.com/BerriAI/litellm&cloudshell_workspace=terraform/litellm/gcp/examples/default&cloudshell_tutorial=tutorial.md)
+
+Opens Cloud Shell (Terraform is pre-installed), clones the repo, and starts a
+guided walkthrough that enables the APIs and runs `terraform apply` against
+your active project.
+
+### AWS — Open in CloudShell
+
+[](https://console.aws.amazon.com/cloudshell/home)
+
+CloudShell already has your AWS credentials. Once it opens, paste:
+
+```bash
+git clone --depth 1 https://github.com/BerriAI/litellm.git
+cd litellm/terraform/litellm/aws/examples/default && ./deploy.sh
+```
+
+`deploy.sh` installs a pinned, checksum-verified Terraform (CloudShell doesn't
+ship one) and applies the stack. Full steps:
+[aws walkthrough](aws/examples/default/tutorial.md).
+
+> **Trial vs. production.** The one-click roots serve **plain HTTP** and
+> register no models — fine for kicking the tires, not for production. For a
+> real deploy, add TLS (`acm_certificate_arn` on AWS, `lb_domains` on GCP),
+> register models via `proxy_config`, and supply your own master key. See the
+> per-stack READMEs.
+
## Components
The proxy is split into three deployables:
| Component | Default image | Port | Role |
| --------- | ---------------------------------------- | ---- | -------------------------------------------------------------------- |
-| `gateway` | `ghcr.io/berriai/litellm-gateway:main-stable` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
-| `backend` | `ghcr.io/berriai/litellm-backend:main-stable` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
-| `ui` | `ghcr.io/berriai/litellm-ui:main-stable` | 3000 | Static Next.js dashboard served by nginx |
+| `gateway` | `ghcr.io/berriai/litellm-gateway:v1.86.0-dev` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
+| `backend` | `ghcr.io/berriai/litellm-backend:v1.86.0-dev` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
+| `ui` | `ghcr.io/berriai/litellm-ui:v1.86.0-dev` | 3000 | Static Next.js dashboard served by nginx |
The load balancer routes gateway path prefixes (mirrored verbatim from
`gateway/routes/allowlist.py`) to the gateway, UI asset paths (`/`,
@@ -136,24 +181,30 @@ everything else to the backend.
## Images
Both stacks take per-component image references as variables. The defaults
-point at the public `ghcr.io/berriai/litellm-:main-stable`
+point at the public `ghcr.io/berriai/litellm-:v1.86.0-dev`
images, so the stack is runnable end-to-end without pre-flight setup —
pin to a specific tag for production:
- **AWS** can pull from any registry the task execution role can reach.
The role gets `AmazonECSTaskExecutionRolePolicy` attached, which grants
- ECR pull permissions for repositories in the same account.
+ ECR pull permissions for repositories in the same account. GHCR is
+ anonymous-readable, so the defaults work as-is.
- **GCP Cloud Run** can only pull from Artifact Registry or
- `gcr.io`-style registries. To use images hosted elsewhere, mirror them
- into Artifact Registry first.
+ `gcr.io`-style registries — it rejects `ghcr.io`. By default the GCP
+ stack auto-creates an Artifact Registry **remote repository** that
+ proxies `https://ghcr.io` (`create_image_proxy_repo = true`), so the
+ upstream images pull with no manual mirroring. Set
+ `create_image_proxy_repo = false` and supply your own `image_registry`
+ to opt out.
## Migrations
LiteLLM's proxy runs `prisma migrate deploy` at startup, but on first apply
the gateway/backend can race the empty database. Both stacks expose a
-one-off migration task that runs `python litellm/proxy/prisma_migration.py`
-against the backend image:
+one-off migration task that runs `python3 /app/run.py` (assembles
+`DATABASE_URL` from the `DATABASE_*` env vars, then `prisma migrate deploy`)
+from the dedicated `ghcr.io/berriai/litellm-migrations` image:
- AWS: an `aws_ecs_task_definition` (`litellm-migrations`). Run with
`aws ecs run-task` — the command is printed in `terraform output`.
@@ -165,11 +216,39 @@ gateway/backend services start serving traffic.
## What's not included
-- TLS certificates / custom domains. Both stacks expose plain-HTTP load
- balancers; bring your own ACM cert (AWS) or managed cert (GCP) and wire
- it into the LB resource.
+- Custom domains / DNS. Both stacks support TLS out of the box — an ACM
+ cert (`acm_certificate_arn`) on AWS, a Google-managed cert (`lb_domains`)
+ on GCP — and `terraform plan` refuses to provision a plaintext LB unless
+ you explicitly opt in (`allow_plaintext_alb` / `allow_plaintext_lb`,
+ which the one-click trial roots default to true). You still bring your
+ own DNS name and point it at the LB; see the per-stack "TLS" sections.
- Remote state backends. Default local state — add an `s3` or `gcs`
backend block to `versions.tf` when graduating to a team environment.
- Observability beyond the cloud provider's defaults (CloudWatch logs on
AWS, Cloud Logging on GCP). Wire your own Prometheus / Datadog / Langfuse
via the `*_extra_env` variables.
+
+## Publishing to the Terraform Registry
+
+These modules are registry-conformant — each is self-contained, declares no
+`provider` block, ships a `README.md` + `examples/default/`, and documents
+every variable/output. The public registry only indexes a module at the
+**root** of a repo named `terraform--`, so the two stacks
+here are mirrored out to dedicated repos rather than published in place:
+
+| Module | Mirror repo | Registry source |
+| ----------------------- | ----------------------------------- | ---------------------- |
+| `terraform/litellm/aws` | `BerriAI/terraform-aws-litellm` | `BerriAI/litellm/aws` |
+| `terraform/litellm/gcp` | `BerriAI/terraform-google-litellm` | `BerriAI/litellm/google` |
+
+The [`Publish Terraform modules`](../../.github/workflows/terraform-modules-publish.yml)
+GitHub Actions workflow does the mirroring: it `git subtree split`s each
+module subdirectory into its mirror repo and tags it with the version you
+pass. Run it manually (Actions → Publish Terraform modules → enter `vX.Y.Z`)
+after a release. One-time setup (create the mirror repos, connect them to the
+registry, add the `TERRAFORM_REGISTRY_SYNC_TOKEN` secret) is documented in
+the workflow header.
+
+Until a version is published to the registry, consume the modules straight
+from this repo with the `github.com/BerriAI/litellm//terraform/litellm/?ref=`
+source shown at the top.
diff --git a/terraform/litellm/aws/README.md b/terraform/litellm/aws/README.md
index 80ee18c667d..c4411526d64 100644
--- a/terraform/litellm/aws/README.md
+++ b/terraform/litellm/aws/README.md
@@ -1,5 +1,22 @@
# LiteLLM on AWS (ECS Fargate)
+[](https://console.aws.amazon.com/cloudshell/home)
+
+> **One-click trial:** open [AWS CloudShell](https://console.aws.amazon.com/cloudshell/home)
+> (your credentials are already there) and paste:
+>
+> ```bash
+> git clone --depth 1 https://github.com/BerriAI/litellm.git
+> cd litellm/terraform/litellm/aws/examples/default && ./deploy.sh
+> ```
+>
+> `deploy.sh` installs a pinned, checksum-verified Terraform and applies the
+> stack. A bare `terraform apply` from `examples/default/` also works with
+> **no tfvars** — it picks the first two AZs in `us-west-2`, serves plain
+> HTTP, and auto-generates a master key. Add `acm_certificate_arn` (TLS) and
+> `proxy_config` (models) for a real deployment. Full steps:
+> [examples/default/tutorial.md](examples/default/tutorial.md).
+
Deploys the componentized LiteLLM proxy on AWS:
- **VPC** with public + private subnets across the AZs you pass in, one NAT gateway
@@ -175,15 +192,24 @@ example files.
## Quick start
+Zero-config trial (no tfvars needed — defaults to `us-west-2`, first two AZs,
+auto-generated master key, HTTP-only):
+
```bash
cd terraform/litellm/aws/examples/default
-cp terraform.tfvars.example terraform.tfvars
-# Edit: region, tenant, env, azs, proxy_config, gateway_extra_secrets.
-
terraform init
terraform apply
```
+To customize, drop in a tfvars file first:
+
+```bash
+cp terraform.tfvars.example terraform.tfvars
+# Optional edits: region, tenant, env, azs, acm_certificate_arn (TLS),
+# proxy_config (models), gateway_extra_secrets (provider keys).
+terraform apply
+```
+
`examples/default/` is a thin root that configures the `aws` provider and
calls the module (`../../`). It exposes a curated variable surface; for
advanced knobs (per-component CPU/memory/workers, autoscaling, RDS/Redis
diff --git a/terraform/litellm/aws/alb.tf b/terraform/litellm/aws/alb.tf
index 786b9d9a5b9..979c48e7273 100644
--- a/terraform/litellm/aws/alb.tf
+++ b/terraform/litellm/aws/alb.tf
@@ -38,6 +38,18 @@ resource "aws_lb_target_group" "gateway" {
deregistration_delay = 30
+ # AWS caps ELB / target-group names at 32 chars. local.name plus the
+ # longest per-resource suffix ("-gateway", 8 chars) must fit, i.e.
+ # length(tenant) + length(env) <= 15 (local.name = "-litellm-").
+ # Checked here so a too-long tenant/env fails at plan with a clear message
+ # instead of an opaque AWS API error deep into apply.
+ lifecycle {
+ precondition {
+ condition = length(local.name) <= 24
+ error_message = "Resource-name prefix '${local.name}' is ${length(local.name)} chars; with an 8-char suffix it exceeds the 32-char AWS load-balancer/target-group limit. Keep length(tenant) + length(env) <= 15."
+ }
+ }
+
tags = local.tags
}
diff --git a/terraform/litellm/aws/ecs.tf b/terraform/litellm/aws/ecs.tf
index aee8f0cfc73..c897c5c8ac0 100644
--- a/terraform/litellm/aws/ecs.tf
+++ b/terraform/litellm/aws/ecs.tf
@@ -135,7 +135,13 @@ locals {
command = [
"python -c \"import os, base64, pathlib; pathlib.Path(os.environ['CONFIG_FILE_PATH']).write_bytes(base64.b64decode(os.environ['LITELLM_PROXY_CONFIG_B64']))\" && exec uvicorn backend.main:app ${local.backend_uvicorn_args}"
]
- } : {}
+ } : {
+ # Pin the backend listener to 0.0.0.0:4001 even without a proxy_config,
+ # so it always matches the target-group health check and container_port.
+ # Mirrors the gateway's no-config branch above.
+ entryPoint = ["uvicorn", "backend.main:app"]
+ command = split(" ", local.backend_uvicorn_args)
+ }
}
# ---------- Gateway ----------
diff --git a/terraform/litellm/aws/examples/default/deploy.sh b/terraform/litellm/aws/examples/default/deploy.sh
new file mode 100755
index 00000000000..b74aebc7e6c
--- /dev/null
+++ b/terraform/litellm/aws/examples/default/deploy.sh
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+#
+# One-command LiteLLM deploy helper for AWS CloudShell (or any machine with
+# the AWS CLI installed and credentials configured).
+#
+# AWS CloudShell ships git + the AWS CLI but not Terraform, so this script
+# installs a pinned, checksum-verified Terraform into ./.bin if it isn't
+# already on PATH, then runs `terraform init` + `terraform apply` against the
+# trial root in this directory.
+#
+# Usage:
+# ./deploy.sh # interactive apply (review plan, type yes)
+# AUTO_APPROVE=1 ./deploy.sh # non-interactive
+#
+# Override the trial defaults with TF_VAR_* env vars or a terraform.tfvars
+# file, e.g.:
+# export TF_VAR_region=us-east-1
+# ./deploy.sh
+set -euo pipefail
+
+TF_VERSION="1.9.8"
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+cd "$SCRIPT_DIR"
+
+# ---- Resolve a terraform binary (install a pinned one if missing) ----
+if command -v terraform >/dev/null 2>&1; then
+ TF="terraform"
+else
+ case "$(uname -m)" in
+ x86_64 | amd64) ARCH="amd64" ;;
+ aarch64 | arm64) ARCH="arm64" ;;
+ *)
+ echo "Unsupported architecture: $(uname -m)" >&2
+ exit 1
+ ;;
+ esac
+
+ BIN_DIR="$SCRIPT_DIR/.bin"
+ TF="$BIN_DIR/terraform"
+ if [ ! -x "$TF" ]; then
+ echo "Installing Terraform ${TF_VERSION} (${ARCH}) into ${BIN_DIR} ..."
+ mkdir -p "$BIN_DIR"
+ tmp="$(mktemp -d)"
+ base="https://releases.hashicorp.com/terraform/${TF_VERSION}"
+ zip="terraform_${TF_VERSION}_linux_${ARCH}.zip"
+ # Download the artifact and HashiCorp's official checksum sidecar, then
+ # verify before unpacking — never trust an unverified download.
+ curl -fsSL -o "$tmp/$zip" "$base/$zip"
+ curl -fsSL -o "$tmp/SHA256SUMS" "$base/terraform_${TF_VERSION}_SHA256SUMS"
+ (cd "$tmp" && grep " $zip\$" SHA256SUMS | sha256sum -c -)
+ unzip -o "$tmp/$zip" -d "$BIN_DIR" >/dev/null
+ rm -rf "$tmp"
+ fi
+fi
+
+echo "Using $("$TF" version | head -1)"
+
+# ---- Sanity-check AWS credentials before spending 15+ minutes applying ----
+if ! aws sts get-caller-identity >/dev/null 2>&1; then
+ echo "AWS credentials not found. In CloudShell this is automatic; otherwise run 'aws configure' or set AWS_PROFILE." >&2
+ exit 1
+fi
+
+# ---- Deploy ----
+"$TF" init -input=false
+if [ "${AUTO_APPROVE:-}" = "1" ]; then
+ "$TF" apply -auto-approve -input=false
+else
+ "$TF" apply -input=false
+fi
+
+echo
+echo "==================================================================="
+echo " LiteLLM is deploying. Useful outputs:"
+echo "==================================================================="
+URL="$("$TF" output -raw alb_url 2>/dev/null || true)"
+KEY_ARN="$("$TF" output -raw master_key_secret_arn 2>/dev/null || true)"
+echo " Proxy URL : ${URL:-}"
+echo " UI login : admin / "
+if [ -n "$KEY_ARN" ]; then
+ echo " Master key: aws secretsmanager get-secret-value --secret-id $KEY_ARN --query SecretString --output text"
+fi
+echo
+echo "The ALB takes a few minutes to pass health checks after apply returns."
diff --git a/terraform/litellm/aws/examples/default/main.tf b/terraform/litellm/aws/examples/default/main.tf
index 0cbd48701aa..ea6f8a0c338 100644
--- a/terraform/litellm/aws/examples/default/main.tf
+++ b/terraform/litellm/aws/examples/default/main.tf
@@ -16,13 +16,24 @@
#
# Knobs not surfaced as variables here (per-component sizing, autoscaling,
# RDS/Redis tuning) can be set directly on this block — see ../../variables.tf.
+
+# When azs is left empty, pick the first two AZs in the region so a
+# zero-config apply works in any region without the caller naming them.
+data "aws_availability_zones" "available" {
+ state = "available"
+}
+
+locals {
+ azs = length(var.azs) > 0 ? var.azs : slice(data.aws_availability_zones.available.names, 0, 2)
+}
+
module "litellm" {
source = "../../"
region = var.region
tenant = var.tenant
env = var.env
- azs = var.azs
+ azs = local.azs
litellm_master_key = var.litellm_master_key
litellm_license = var.litellm_license
diff --git a/terraform/litellm/aws/examples/default/terraform.tfvars.example b/terraform/litellm/aws/examples/default/terraform.tfvars.example
index 88d12cf26e2..2caeee1223b 100644
--- a/terraform/litellm/aws/examples/default/terraform.tfvars.example
+++ b/terraform/litellm/aws/examples/default/terraform.tfvars.example
@@ -1,12 +1,19 @@
-region = "us-west-2"
-azs = ["us-west-2a", "us-west-2b"]
+# EVERYTHING in this file is optional. With no tfvars at all, `terraform
+# apply` brings up a working HTTP-only trial instance: region defaults to
+# us-west-2, the first two AZs in the region are picked automatically,
+# tenant/env default to litellm/trial, and a master key is auto-generated.
+# Override any of the values below for a real deployment.
+
+# region = "us-west-2"
+# azs = ["us-west-2a", "us-west-2b"] # empty → first two AZs in the region
# Resource naming: every AWS resource the stack creates is named
# `${tenant}-litellm-${env}` (or that plus a per-resource suffix). E.g.
# tenant="acme" + env="stage" → ALB `acme-litellm-stage`, ECS service
-# `acme-litellm-stage-gateway`, etc.
-tenant = "acme"
-env = "stage"
+# `acme-litellm-stage-gateway`, etc. Keep length(tenant)+length(env) <= 15
+# (AWS caps ELB/target-group names at 32 chars).
+# tenant = "acme"
+# env = "stage"
# Tenant-supplied secrets. Prefer TF_VAR_litellm_master_key /
# TF_VAR_litellm_license / TF_VAR_ui_password env vars so the values don't
@@ -17,10 +24,10 @@ env = "stage"
# litellm_license = "lic-..."
# ui_password = "..."
-# TLS: provide an ACM cert for production. Without one, plan fails unless
-# allow_plaintext_alb = true is set explicitly (trial/dev only).
+# TLS: this trial root defaults to HTTP-only (allow_plaintext_alb = true).
+# For a real deployment, provide an ACM cert and turn plaintext back off.
# acm_certificate_arn = "arn:aws:acm:us-west-2:111122223333:certificate/..."
-# allow_plaintext_alb = true
+# allow_plaintext_alb = false
# Storage retention: false (default) makes `terraform destroy` refuse on a
# non-empty bucket. Flip to true only for ephemeral / CI stacks.
diff --git a/terraform/litellm/aws/examples/default/tutorial.md b/terraform/litellm/aws/examples/default/tutorial.md
new file mode 100644
index 00000000000..05fdddb36de
--- /dev/null
+++ b/terraform/litellm/aws/examples/default/tutorial.md
@@ -0,0 +1,80 @@
+# Deploy LiteLLM on AWS
+
+This walkthrough deploys the **LiteLLM AI Gateway** into your own AWS account
+using Terraform. You get the full componentized proxy — gateway, backend, and
+dashboard on ECS Fargate, fronted by an Application Load Balancer, backed by
+Aurora Postgres, ElastiCache (Redis), and an S3 bucket.
+
+The fastest path is **AWS CloudShell**, which already has the AWS CLI and your
+credentials wired up. Click the button in the
+[module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/aws/README.md),
+or open and follow along.
+
+## 1. Get the code
+
+In CloudShell (or any machine with the AWS CLI configured):
+
+```bash
+git clone --depth 1 https://github.com/BerriAI/litellm.git
+cd litellm/terraform/litellm/aws/examples/default
+```
+
+## 2. Deploy
+
+`deploy.sh` installs a pinned, checksum-verified Terraform (CloudShell doesn't
+ship one), then runs `terraform init` + `terraform apply`:
+
+```bash
+./deploy.sh
+```
+
+Review the plan and type `yes`. The apply provisions the VPC, Aurora cluster,
+Redis, S3, and the three ECS services, bootstraps the database, runs the
+schema migration, and only then starts the services — so it takes **15-20
+minutes** on the first run.
+
+> Prefer to drive Terraform yourself? `terraform init && terraform apply` works
+> too, as long as Terraform is already installed.
+
+## 3. You're live
+
+```bash
+terraform output alb_url
+```
+
+The dashboard is at `/`, the OpenAI-compatible API at `/v1/*`. Log in with
+username `admin` and the auto-generated master key:
+
+```bash
+aws secretsmanager get-secret-value \
+ --secret-id "$(terraform output -raw master_key_secret_arn)" \
+ --query SecretString --output text
+```
+
+The ALB takes a few minutes to pass health checks after apply returns.
+
+## 4. Customize (optional)
+
+This trial deploy serves plain HTTP and registers no models. For a real
+deployment, copy and edit the tfvars file, then re-apply:
+
+```bash
+cp terraform.tfvars.example terraform.tfvars
+# Edit: region, tenant/env, acm_certificate_arn (TLS), proxy_config (models),
+# gateway_extra_secrets (provider API keys).
+terraform apply
+```
+
+Provider API keys go in AWS Secrets Manager and are referenced by ARN — see
+the [module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/aws/README.md)
+for the full configuration surface (TLS, models, sizing, multi-tenant).
+
+## 5. Clean up
+
+```bash
+terraform destroy
+```
+
+Aurora takes a final snapshot and the S3 bucket refuses to delete while
+non-empty (data-loss guards). Set `skip_final_snapshot = true` /
+`s3_force_destroy = true` for an ephemeral trial you don't mind losing.
diff --git a/terraform/litellm/aws/examples/default/variables.tf b/terraform/litellm/aws/examples/default/variables.tf
index f8950ca2eca..85436b0d3c4 100644
--- a/terraform/litellm/aws/examples/default/variables.tf
+++ b/terraform/litellm/aws/examples/default/variables.tf
@@ -5,24 +5,30 @@
# per-variable docs live in ../../variables.tf — the module is the source
# of truth; descriptions here are intentionally terse.
+# Defaults make a bare `terraform apply` bring up a working trial instance
+# (no tfvars required). Override any of them for a real deployment.
variable "region" {
description = "AWS region to deploy into."
type = string
+ default = "us-west-2"
}
variable "tenant" {
description = "Tenant slug — prefix for every resource (-litellm-)."
type = string
+ default = "litellm"
}
variable "env" {
description = "Environment suffix (stage, prod, dev)."
type = string
+ default = "trial"
}
variable "azs" {
- description = "Availability zones for subnets. At least 2 (RDS + ALB)."
+ description = "Availability zones for subnets. At least 2 (RDS + ALB). Empty (default) auto-picks the first two AZs in the region."
type = list(string)
+ default = []
}
# Sensitive — prefer TF_VAR_litellm_master_key / TF_VAR_litellm_license /
@@ -56,9 +62,9 @@ variable "acm_certificate_arn" {
}
variable "allow_plaintext_alb" {
- description = "Opt into HTTP-only ALB (trial/dev only)."
+ description = "Opt into HTTP-only ALB (trial/dev only). Defaults true in this trial root so a zero-config apply succeeds; set acm_certificate_arn (and flip this to false) for a real deployment."
type = bool
- default = false
+ default = true
}
variable "s3_force_destroy" {
diff --git a/terraform/litellm/gcp/README.md b/terraform/litellm/gcp/README.md
index 140741bcff6..67d5efa4d90 100644
--- a/terraform/litellm/gcp/README.md
+++ b/terraform/litellm/gcp/README.md
@@ -1,5 +1,14 @@
# LiteLLM on GCP (Cloud Run)
+[](https://shell.cloud.google.com/cloudshell/editor?cloudshell_git_repo=https://github.com/BerriAI/litellm&cloudshell_workspace=terraform/litellm/gcp/examples/default&cloudshell_tutorial=tutorial.md)
+
+> **One-click trial:** the button above opens Cloud Shell with a guided
+> walkthrough that deploys this stack into your active project. A bare
+> `terraform apply` from `examples/default/` also works with **no tfvars** —
+> it serves plain HTTP, auto-generates a master key, and auto-creates an
+> Artifact Registry proxy so Cloud Run can pull the images. Add `lb_domains`
+> (TLS) and `proxy_config` (models) for a real deployment.
+
Deploys the componentized LiteLLM proxy on GCP:
- **VPC** + Private Services Access range + a Serverless VPC Access connector
@@ -27,11 +36,21 @@ Bump them together when bumping LiteLLM.
Cloud Run only accepts images from Artifact Registry, `[region.]gcr.io`,
or `docker.io` — `ghcr.io` URIs are rejected at apply time. The four
-images are published to GHCR upstream, so any real deploy needs an
-Artifact Registry remote repository pointed at GHCR.
+images are published to GHCR upstream, so a remote Artifact Registry
+repository is needed to proxy them.
-**One-time setup (per project):** create a remote repo and let Cloud Run
-pull through it.
+**By default the stack creates this for you.** With
+`create_image_proxy_repo = true` (the default), Terraform provisions an
+Artifact Registry remote repository pointed at `https://ghcr.io`, grants
+the project's serverless agent read on it, and — when `image_registry` is
+left empty — composes the four image URIs from it automatically:
+`-docker.pkg.dev//-litellm--ghcr/berriai/litellm-:`.
+That's what makes the zero-config deploy work end-to-end with no manual
+mirroring step.
+
+**To use your own registry instead**, set `create_image_proxy_repo = false`
+and point `image_registry` at an existing Artifact Registry path. The
+manual equivalent of the auto-created repo is:
```bash
gcloud artifacts repositories create litellm \
@@ -42,11 +61,10 @@ gcloud artifacts repositories create litellm \
--remote-docker-repo=https://ghcr.io
```
-Then point the stack at it via `image_registry`:
-
```hcl
-image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
-image_tag = "v1.86.0-dev"
+create_image_proxy_repo = false
+image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
+image_tag = "v1.86.0-dev"
```
The four `litellm-:${image_tag}` URIs are composed from those
@@ -54,15 +72,8 @@ two vars. Set `gateway_image` / `backend_image` / `ui_image` /
`migrations_image` only if you need a per-component override (custom
build, different tag).
-Two further notes:
-
-- The runtime SAs the stack creates do **not** need
- `roles/artifactregistry.reader` — Cloud Run pulls images using the
- per-project serverless agent
- (`service-@serverless-robot-prod.iam.gserviceaccount.com`),
- not the runtime SA.
-- For a fully air-gapped option, mirror the images into a regular AR
- repository instead of a remote repo:
+For a fully air-gapped option, mirror the images into a regular AR
+repository instead of a remote repo:
```bash
for c in gateway backend ui migrations; do
@@ -204,15 +215,24 @@ example files.
## Quick start
+Zero-config trial (no tfvars needed — project is inferred from your active
+gcloud/ADC project):
+
```bash
cd terraform/litellm/gcp/examples/default
-cp terraform.tfvars.example terraform.tfvars
-# Edit: project, region, tenant, env, image_registry, proxy_config, gateway_extra_secrets.
-
terraform init
terraform apply
```
+To customize, drop in a tfvars file first:
+
+```bash
+cp terraform.tfvars.example terraform.tfvars
+# Optional edits: project, region, tenant, env, lb_domains (TLS),
+# proxy_config (models), gateway_extra_secrets (provider keys).
+terraform apply
+```
+
`examples/default/` is a thin root that configures the `google` /
`google-beta` providers and calls the module (`../../`). It exposes a
curated variable surface; for advanced knobs (per-component
diff --git a/terraform/litellm/gcp/artifact_registry.tf b/terraform/litellm/gcp/artifact_registry.tf
new file mode 100644
index 00000000000..5262c701f8c
--- /dev/null
+++ b/terraform/litellm/gcp/artifact_registry.tf
@@ -0,0 +1,49 @@
+# Cloud Run can only pull from Artifact Registry, [region.]gcr.io, or
+# docker.io — it rejects ghcr.io URIs at apply time. The four LiteLLM
+# images live on GHCR upstream, so by default this stack provisions an
+# Artifact Registry *remote repository* that transparently proxies
+# https://ghcr.io. Cloud Run then pulls `…-docker.pkg.dev///
+# berriai/litellm-` and AR fetches+caches from GHCR on first pull.
+#
+# This is what makes a zero-config deploy possible: with the proxy in place
+# the default `image_registry` resolves to the proxy path (see locals.tf),
+# so no manual `gcloud artifacts repositories create` step is needed.
+#
+# Set create_image_proxy_repo = false (and supply your own image_registry /
+# *_image) to skip it — e.g. when mirroring images into a standard AR repo.
+
+data "google_project" "this" {
+ project_id = var.project
+}
+
+resource "google_artifact_registry_repository" "ghcr_proxy" {
+ count = var.create_image_proxy_repo ? 1 : 0
+
+ location = var.region
+ repository_id = "${local.name}-ghcr"
+ description = "GitHub Container Registry (ghcr.io) passthrough for LiteLLM images"
+ format = "DOCKER"
+ mode = "REMOTE_REPOSITORY"
+ labels = var.labels
+
+ remote_repository_config {
+ description = "ghcr.io"
+ docker_repository {
+ custom_repository {
+ uri = "https://ghcr.io"
+ }
+ }
+ }
+}
+
+# Cloud Run pulls images with the per-project serverless service agent, not
+# the runtime SA. Grant that agent read on the proxy repo so the pull (and
+# the upstream fetch) succeeds.
+resource "google_artifact_registry_repository_iam_member" "serverless_agent_reader" {
+ count = var.create_image_proxy_repo ? 1 : 0
+
+ location = google_artifact_registry_repository.ghcr_proxy[0].location
+ repository = google_artifact_registry_repository.ghcr_proxy[0].name
+ role = "roles/artifactregistry.reader"
+ member = "serviceAccount:service-${data.google_project.this.number}@serverless-robot-prod.iam.gserviceaccount.com"
+}
diff --git a/terraform/litellm/gcp/cloudrun.tf b/terraform/litellm/gcp/cloudrun.tf
index 28e1145b081..d5e671de37c 100644
--- a/terraform/litellm/gcp/cloudrun.tf
+++ b/terraform/litellm/gcp/cloudrun.tf
@@ -117,6 +117,16 @@ resource "google_cloud_run_v2_service" "gateway" {
name = "${local.name}-gateway"
location = var.region
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
+ labels = var.labels
+
+ # Cloud Run rejects ghcr.io images. Catch the one misconfiguration that
+ # otherwise fails deep into apply: registry cleared AND proxy disabled.
+ lifecycle {
+ precondition {
+ condition = !startswith(local.gateway_image, "ghcr.io/") && !startswith(local.backend_image, "ghcr.io/") && !startswith(local.ui_image, "ghcr.io/") && !startswith(local.migrations_image, "ghcr.io/")
+ error_message = "Cloud Run cannot pull from ghcr.io. Keep create_image_proxy_repo = true (default) to auto-create an Artifact Registry remote repo, or set image_registry to an Artifact Registry path."
+ }
+ }
template {
service_account = google_service_account.runtime.email
@@ -197,6 +207,9 @@ resource "google_cloud_run_v2_service" "gateway" {
google_secret_manager_secret_iam_member.license,
google_secret_manager_secret_iam_member.extras,
google_sql_user.app,
+ # The serverless agent needs read on the image-proxy repo before the
+ # first pull (no-op when create_image_proxy_repo = false).
+ google_artifact_registry_repository_iam_member.serverless_agent_reader,
# Don't go live until the schema is migrated; otherwise the proxy boots,
# fails on missing tables, and Cloud Run keeps cold-restarting.
terraform_data.migration,
@@ -208,6 +221,7 @@ resource "google_cloud_run_v2_service" "backend" {
name = "${local.name}-backend"
location = var.region
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
+ labels = var.labels
template {
service_account = google_service_account.runtime.email
@@ -289,6 +303,7 @@ resource "google_cloud_run_v2_service" "backend" {
google_secret_manager_secret_iam_member.ui_password,
google_secret_manager_secret_iam_member.extras,
google_sql_user.app,
+ google_artifact_registry_repository_iam_member.serverless_agent_reader,
terraform_data.migration,
]
}
@@ -301,6 +316,7 @@ resource "google_cloud_run_v2_service" "ui" {
name = "${local.name}-ui"
location = var.region
ingress = "INGRESS_TRAFFIC_INTERNAL_LOAD_BALANCER"
+ labels = var.labels
template {
service_account = google_service_account.ui_runtime.email
@@ -337,6 +353,10 @@ resource "google_cloud_run_v2_service" "ui" {
}
}
}
+
+ depends_on = [
+ google_artifact_registry_repository_iam_member.serverless_agent_reader,
+ ]
}
# Allow the LB (any unauthenticated traffic from the configured serverless
@@ -374,6 +394,7 @@ resource "google_cloud_run_v2_service_iam_member" "ui_allusers" {
resource "google_cloud_run_v2_job" "migrations" {
name = "${local.name}-migrations"
location = var.region
+ labels = var.labels
template {
template {
@@ -426,5 +447,6 @@ resource "google_cloud_run_v2_job" "migrations" {
depends_on = [
google_secret_manager_secret_iam_member.db_password,
google_sql_user.app,
+ google_artifact_registry_repository_iam_member.serverless_agent_reader,
]
}
diff --git a/terraform/litellm/gcp/cloudsql.tf b/terraform/litellm/gcp/cloudsql.tf
index e3394fefc0f..0126ff4e013 100644
--- a/terraform/litellm/gcp/cloudsql.tf
+++ b/terraform/litellm/gcp/cloudsql.tf
@@ -25,6 +25,7 @@ resource "google_sql_database_instance" "writer" {
availability_type = "REGIONAL"
disk_size = 20
disk_autoresize = true
+ user_labels = var.labels
backup_configuration {
enabled = true
@@ -69,6 +70,7 @@ resource "google_sql_database_instance" "reader" {
tier = var.db_tier
availability_type = "ZONAL"
disk_autoresize = true
+ user_labels = var.labels
ip_configuration {
ipv4_enabled = false
diff --git a/terraform/litellm/gcp/examples/default/main.tf b/terraform/litellm/gcp/examples/default/main.tf
index 745b79383db..a4c51fd214e 100644
--- a/terraform/litellm/gcp/examples/default/main.tf
+++ b/terraform/litellm/gcp/examples/default/main.tf
@@ -17,10 +17,20 @@
# Knobs not surfaced as variables here (per-component sizing/instances,
# Cloud SQL tier/edition, Memorystore tier, per-component image overrides)
# can be set directly on this block — see ../../variables.tf.
+
+# Resolve the effective project: explicit var.project wins, otherwise fall
+# back to whatever the provider inferred from gcloud / ADC (Cloud Shell sets
+# this). The module needs a concrete project ID for project-scoped IAM.
+data "google_client_config" "current" {}
+
+locals {
+ project = var.project != "" ? var.project : data.google_client_config.current.project
+}
+
module "litellm" {
source = "../../"
- project = var.project
+ project = local.project
region = var.region
tenant = var.tenant
env = var.env
diff --git a/terraform/litellm/gcp/examples/default/providers.tf b/terraform/litellm/gcp/examples/default/providers.tf
index 4b79367fe09..cee5d45ddfb 100644
--- a/terraform/litellm/gcp/examples/default/providers.tf
+++ b/terraform/litellm/gcp/examples/default/providers.tf
@@ -6,12 +6,16 @@
# The module's resources inherit these default (unaliased) `google` /
# `google-beta` configs automatically through the module call, so project
# and region set here flow into every resource that doesn't pass its own.
+# project = null when var.project is empty, which lets the provider infer
+# the project from the active gcloud config / ADC (set in Cloud Shell). The
+# resolved value is read back via data.google_client_config in main.tf and
+# passed explicitly to the module.
provider "google" {
- project = var.project
+ project = var.project != "" ? var.project : null
region = var.region
}
provider "google-beta" {
- project = var.project
+ project = var.project != "" ? var.project : null
region = var.region
}
diff --git a/terraform/litellm/gcp/examples/default/terraform.tfvars.example b/terraform/litellm/gcp/examples/default/terraform.tfvars.example
index eff338ca240..6dca464eb44 100644
--- a/terraform/litellm/gcp/examples/default/terraform.tfvars.example
+++ b/terraform/litellm/gcp/examples/default/terraform.tfvars.example
@@ -1,12 +1,19 @@
-project = "my-gcp-project"
-region = "us-central1"
+# EVERYTHING in this file is optional. With no tfvars at all, `terraform
+# apply` brings up a working HTTP-only trial instance: project is inferred
+# from your active gcloud/ADC project, region defaults to us-central1,
+# tenant/env default to litellm/trial, and the module auto-creates an
+# Artifact Registry proxy so Cloud Run can pull the ghcr.io images. Override
+# any of the values below for a real deployment.
+
+# project = "my-gcp-project" # empty → inferred from gcloud / ADC
+# region = "us-central1"
# Resource naming: every GCP resource the stack creates is named
# `${tenant}-litellm-${env}` (or that plus a per-resource suffix). E.g.
# tenant="acme" + env="stage" → Cloud Run service `acme-litellm-stage-gateway`,
# Cloud SQL instance `acme-litellm-stage`, etc.
-tenant = "acme"
-env = "stage"
+# tenant = "acme"
+# env = "stage"
# Tenant-supplied secrets. Prefer TF_VAR_litellm_master_key /
# TF_VAR_litellm_license / TF_VAR_ui_password env vars so the values don't
@@ -17,23 +24,23 @@ env = "stage"
# litellm_license = "lic-..."
# ui_password = "..."
-# TLS: provide DNS names already pointing at the LB IP for a Google-managed
-# cert. Without one, plan fails unless allow_plaintext_lb = true is set
-# explicitly (trial/dev only).
+# TLS: this trial root defaults to HTTP-only (allow_plaintext_lb = true).
+# For a real deployment, provide DNS names already pointing at the LB IP for
+# a Google-managed cert and turn plaintext back off.
# lb_domains = ["proxy.example.com"]
-# allow_plaintext_lb = true
+# allow_plaintext_lb = false
# Storage and database retention. Defaults are safe — destroy preserves
# data. Flip these only for ephemeral / CI stacks.
# cloudsql_deletion_protection = true # default: refuse destroy on the DB
# gcs_force_destroy = false # default: refuse destroy on a non-empty bucket
-# Images. Cloud Run rejects ghcr.io, so a real deploy must point
-# image_registry at an Artifact Registry remote repo (see README "Image
-# pulls"); image_tag is applied to all four litellm-* images. Per-component
-# *_image overrides are NOT exposed here — set them directly on the
-# `module "litellm"` block in main.tf (see ../../variables.tf) if you need
-# to mix-and-match versions.
+# Images. Left empty, the module auto-creates an Artifact Registry remote
+# repo proxying ghcr.io (Cloud Run rejects ghcr.io directly), so the default
+# images pull with no setup. Point image_registry at your own Artifact
+# Registry repo to bypass the proxy; image_tag applies to all four litellm-*
+# images. Per-component *_image overrides are NOT exposed here — set them on
+# the `module "litellm"` block in main.tf (see ../../variables.tf).
# image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
# image_tag = "v1.86.0-dev"
diff --git a/terraform/litellm/gcp/examples/default/tutorial.md b/terraform/litellm/gcp/examples/default/tutorial.md
new file mode 100644
index 00000000000..d379023e2a5
--- /dev/null
+++ b/terraform/litellm/gcp/examples/default/tutorial.md
@@ -0,0 +1,131 @@
+# Deploy LiteLLM on Google Cloud
+
+
+
+This guided walkthrough deploys the **LiteLLM AI Gateway** into your own
+Google Cloud project using Terraform. You get the full componentized proxy —
+gateway, backend, and dashboard on Cloud Run, fronted by an external HTTP(S)
+load balancer, backed by Cloud SQL (Postgres), Memorystore (Redis), and a
+GCS bucket.
+
+Everything runs from this Cloud Shell session — Terraform is already
+installed here, so there's nothing to set up on your machine.
+
+## Choose your project
+
+Pick the Google Cloud project to deploy into. Everything the stack creates is
+billed to and lives in this project.
+
+
+
+Set it as the active project for this session:
+
+```bash
+gcloud config set project
+```
+
+## Enable the required APIs
+
+The stack uses Cloud Run, Cloud SQL, Memorystore, Secret Manager, Serverless
+VPC Access, Compute, Service Networking, Cloud Storage, and Artifact Registry.
+Enable them all in one call (this can take a minute):
+
+```bash
+gcloud services enable \
+ run.googleapis.com \
+ sqladmin.googleapis.com \
+ redis.googleapis.com \
+ secretmanager.googleapis.com \
+ vpcaccess.googleapis.com \
+ compute.googleapis.com \
+ servicenetworking.googleapis.com \
+ storage.googleapis.com \
+ artifactregistry.googleapis.com
+```
+
+## Deploy
+
+Tell Terraform which project to use, then initialize and apply. The apply
+provisions everything, runs the database migration, and only then starts the
+services — so it takes **15-20 minutes** on the first run (Cloud SQL and the
+load balancer are the slow parts).
+
+```bash
+export TF_VAR_project=$(gcloud config get-value project)
+terraform init
+terraform apply
+```
+
+Review the plan and type `yes` to proceed.
+
+This trial deploy serves plain HTTP and auto-generates
+a master key. For a production deploy, set `lb_domains` for a managed TLS
+cert and supply your own master key — see the module README.
+
+## You're live
+
+Print the proxy URL (the dashboard is at `/`, the OpenAI-compatible API at
+`/v1/*`):
+
+```bash
+terraform output lb_url
+```
+
+Fetch the auto-generated admin / master key — log into the dashboard with
+username `admin` and this value:
+
+```bash
+gcloud secrets versions access latest \
+ --secret="$(terraform output -raw master_key_secret_id)"
+```
+
+Send a test request (replace `URL` and `KEY` with the two values above):
+
+```bash
+curl "$(terraform output -raw lb_url)/v1/models" \
+ -H "Authorization: Bearer $(gcloud secrets versions access latest --secret="$(terraform output -raw master_key_secret_id)")"
+```
+
+## Add a model
+
+Edit `terraform.tfvars` to register models and provider keys, then re-apply.
+Store provider API keys in Secret Manager and reference them, e.g.:
+
+```bash
+echo -n "sk-proj-..." | gcloud secrets create openai-api-key --data-file=-
+```
+
+```hcl
+proxy_config = {
+ model_list = [{
+ model_name = "gpt-4o"
+ litellm_params = { model = "openai/gpt-4o", api_key = "os.environ/OPENAI_API_KEY" }
+ }]
+}
+gateway_extra_secrets = {
+ OPENAI_API_KEY = "projects//secrets/openai-api-key"
+}
+```
+
+Then `terraform apply` again.
+
+## Clean up
+
+To tear everything down when you're done:
+
+```bash
+terraform destroy
+```
+
+Cloud SQL has deletion protection on by default, so
+destroy will refuse until you set `cloudsql_deletion_protection = false` (and
+`gcs_force_destroy = true` for a non-empty bucket) and re-apply. That's a
+guard against accidental data loss.
+
+## Done
+
+
+
+You've deployed LiteLLM on Google Cloud. For configuration options (models,
+TLS, sizing, multi-tenant deploys), see the
+[module README](https://github.com/BerriAI/litellm/blob/main/terraform/litellm/gcp/README.md).
diff --git a/terraform/litellm/gcp/examples/default/variables.tf b/terraform/litellm/gcp/examples/default/variables.tf
index 745a5e5d76b..a2987ceac0a 100644
--- a/terraform/litellm/gcp/examples/default/variables.tf
+++ b/terraform/litellm/gcp/examples/default/variables.tf
@@ -5,9 +5,13 @@
# main.tf, or call the module from your own root config. Full per-variable
# docs live in ../../variables.tf — the module is the source of truth.
+# Defaults make a bare `terraform apply` bring up a working trial instance.
+# `project` is the one value that has no safe default — empty means "infer
+# from the active gcloud/ADC project" (which Cloud Shell sets for you).
variable "project" {
- description = "GCP project ID."
+ description = "GCP project ID. Empty (default) infers the active gcloud / ADC project (set automatically in Cloud Shell)."
type = string
+ default = ""
}
variable "region" {
@@ -19,11 +23,13 @@ variable "region" {
variable "tenant" {
description = "Tenant slug — prefix for every resource (-litellm-)."
type = string
+ default = "litellm"
}
variable "env" {
description = "Environment suffix (stage, prod, dev)."
type = string
+ default = "trial"
}
# Sensitive — prefer TF_VAR_litellm_master_key / TF_VAR_litellm_license /
@@ -49,13 +55,14 @@ variable "ui_password" {
sensitive = true
}
-# Image source. Cloud Run rejects ghcr.io, so a real deploy must point
-# image_registry at an Artifact Registry remote repo (see README "Image
-# pulls"). Per-component overrides live in ../../variables.tf.
+# Image source. Empty (default) makes the module auto-create an Artifact
+# Registry remote repo proxying ghcr.io (Cloud Run rejects ghcr.io directly),
+# so images pull with no manual setup. Set this to your own Artifact Registry
+# path to bypass the proxy. Per-component overrides live in ../../variables.tf.
variable "image_registry" {
- description = "Registry path prefix; images composed as /litellm-:."
+ description = "Registry path prefix; images composed as /litellm-:. Empty → auto ghcr.io proxy repo."
type = string
- default = "ghcr.io/berriai"
+ default = ""
}
variable "image_tag" {
@@ -72,9 +79,9 @@ variable "lb_domains" {
}
variable "allow_plaintext_lb" {
- description = "Opt into HTTP-only LB (trial/dev only)."
+ description = "Opt into HTTP-only LB (trial/dev only). Defaults true in this trial root so a zero-config apply succeeds; set lb_domains (and flip this to false) for a real deployment."
type = bool
- default = false
+ default = true
}
variable "cloudsql_deletion_protection" {
diff --git a/terraform/litellm/gcp/locals.tf b/terraform/litellm/gcp/locals.tf
index 2d1231fb197..2824cbbf071 100644
--- a/terraform/litellm/gcp/locals.tf
+++ b/terraform/litellm/gcp/locals.tf
@@ -69,11 +69,22 @@ locals {
{ name = "CONFIG_FILE_PATH", value = "/tmp/litellm-config.yaml" },
] : []
+ # Effective registry prefix. An explicit image_registry wins; otherwise,
+ # when the ghcr.io proxy repo is created, images resolve to it (mirrors the
+ # upstream `berriai/litellm-*` path under the remote repo). Falling all the
+ # way through to ghcr.io is only reachable when the operator both clears
+ # image_registry and disables the proxy — Cloud Run rejects it, so the
+ # per-service precondition (cloudrun.tf) fails fast with guidance.
+ image_registry = (
+ var.image_registry != "" ? var.image_registry :
+ var.create_image_proxy_repo ? "${var.region}-docker.pkg.dev/${var.project}/${google_artifact_registry_repository.ghcr_proxy[0].repository_id}/berriai" :
+ "ghcr.io/berriai"
+ )
+
# Resolved image URIs: per-component override wins, otherwise compose
- # from image_registry + image_tag. Cloud Run only accepts AR / gcr.io /
- # docker.io paths — see variables.tf for the full constraint list.
- gateway_image = var.gateway_image != "" ? var.gateway_image : "${var.image_registry}/litellm-gateway:${var.image_tag}"
- backend_image = var.backend_image != "" ? var.backend_image : "${var.image_registry}/litellm-backend:${var.image_tag}"
- ui_image = var.ui_image != "" ? var.ui_image : "${var.image_registry}/litellm-ui:${var.image_tag}"
- migrations_image = var.migrations_image != "" ? var.migrations_image : "${var.image_registry}/litellm-migrations:${var.image_tag}"
+ # from the effective registry + image_tag.
+ gateway_image = var.gateway_image != "" ? var.gateway_image : "${local.image_registry}/litellm-gateway:${var.image_tag}"
+ backend_image = var.backend_image != "" ? var.backend_image : "${local.image_registry}/litellm-backend:${var.image_tag}"
+ ui_image = var.ui_image != "" ? var.ui_image : "${local.image_registry}/litellm-ui:${var.image_tag}"
+ migrations_image = var.migrations_image != "" ? var.migrations_image : "${local.image_registry}/litellm-migrations:${var.image_tag}"
}
diff --git a/terraform/litellm/gcp/redis.tf b/terraform/litellm/gcp/redis.tf
index f7e174ecbae..095ccdde645 100644
--- a/terraform/litellm/gcp/redis.tf
+++ b/terraform/litellm/gcp/redis.tf
@@ -3,6 +3,7 @@ resource "google_redis_instance" "this" {
tier = var.redis_tier
memory_size_gb = var.redis_memory_size_gb
region = var.region
+ labels = var.labels
authorized_network = google_compute_network.this.id
connect_mode = "PRIVATE_SERVICE_ACCESS"
diff --git a/terraform/litellm/gcp/variables.tf b/terraform/litellm/gcp/variables.tf
index fe726b0317a..d6ca8c01113 100644
--- a/terraform/litellm/gcp/variables.tf
+++ b/terraform/litellm/gcp/variables.tf
@@ -30,7 +30,7 @@ variable "env" {
}
variable "labels" {
- description = "Resource labels merged into every label-supporting resource."
+ description = "Resource labels applied to the billable / filterable resources: the three Cloud Run services, the migrations job, Cloud SQL (writer + reader), Memorystore, the GCS bucket, and the image-proxy Artifact Registry repo. (Compute networking and Secret Manager resources don't carry labels.)"
type = map(string)
default = {
"managed-by" = "terraform"
@@ -104,19 +104,35 @@ variable "vpc_connector_cidr" {
# pointed at ghcr.io (e.g. `us-central1-docker.pkg.dev/my-proj/litellm/berriai`)
# or override the per-component `*_image` vars individually with full URIs.
+variable "create_image_proxy_repo" {
+ description = <<-EOT
+ Create an Artifact Registry remote repository that proxies
+ `https://ghcr.io`, so Cloud Run (which rejects ghcr.io URIs) can pull
+ the upstream LiteLLM images without a manual mirroring step. Default
+ true — this is what lets a zero-config deploy run end-to-end. When true
+ and `image_registry` is left empty, the four image URIs resolve to the
+ proxy repo automatically (see locals.tf). Set false if you supply your
+ own `image_registry` / `*_image` (e.g. a standard AR repo you mirror
+ into). Requires the `artifactregistry.googleapis.com` API.
+ EOT
+ type = bool
+ default = true
+}
+
variable "image_registry" {
description = <<-EOT
Registry path prefix used to compose the four LiteLLM image URIs as
- `/litellm-:`. The default
- (`ghcr.io/berriai`) only works on registries Cloud Run accepts — for
- GHCR-backed deploys, create an Artifact Registry remote repository
- pointed at `https://ghcr.io` and set this to that repo's path
- (e.g. `us-central1-docker.pkg.dev///berriai`).
+ `/litellm-:`. Empty (default)
+ resolves to the auto-created ghcr.io proxy repo when
+ `create_image_proxy_repo = true`; otherwise it falls back to
+ `ghcr.io/berriai` (which Cloud Run rejects — a precondition catches
+ this). For a custom registry, set this to an Artifact Registry path
+ (e.g. `us-central1-docker.pkg.dev///berriai`).
Per-component overrides (`gateway_image`, `backend_image`, `ui_image`,
`migrations_image`) bypass this entirely when set.
EOT
type = string
- default = "ghcr.io/berriai"
+ default = ""
}
variable "image_tag" {