mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge b4ac589345 into 9eac594850
This commit is contained in:
commit
4767767953
24 changed files with 404 additions and 359 deletions
|
|
@ -34,11 +34,16 @@ secret-manager refs.
|
|||
|
||||
The proxy is split into three deployables:
|
||||
|
||||
| Component | Default image | Port | Role |
|
||||
| --------- | ---------------------------------------- | ---- | -------------------------------------------------------------------- |
|
||||
| `gateway` | `ghcr.io/berriai/litellm-gateway:main-stable` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
|
||||
| `backend` | `ghcr.io/berriai/litellm-backend:main-stable` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
|
||||
| `ui` | `ghcr.io/berriai/litellm-ui:main-stable` | 3000 | Static Next.js dashboard served by nginx |
|
||||
| Component | Entrypoint argument | Port | Role |
|
||||
| --------- | ------------------- | ---- | -------------------------------------------------------------------- |
|
||||
| `gateway` | `gateway` | 4000 | LLM data plane (`/v1/chat/completions`, `/v1/embeddings`, …) |
|
||||
| `backend` | `backend` | 4001 | Management API (`/key/*`, `/user/*`, `/team/*`, `/model/*`, …) |
|
||||
| `ui` | `ui` | 3000 | Static Next.js dashboard served by nginx |
|
||||
|
||||
All of them run the same `ghcr.io/berriai/litellm` image (`image` on AWS,
|
||||
`image_registry` + `image_tag` or `image` on GCP). The image entrypoint
|
||||
starts the process named by its first argument, so each workload only
|
||||
differs in the argument the module passes
|
||||
|
||||
The load balancer routes gateway path prefixes (mirrored verbatim from
|
||||
`gateway/routes/allowlist.py`) to the gateway, UI asset paths (`/`,
|
||||
|
|
@ -152,8 +157,8 @@ pin to a specific tag for production:
|
|||
|
||||
LiteLLM's proxy runs `prisma migrate deploy` at startup, but on first apply
|
||||
the gateway/backend can race the empty database. Both stacks expose a
|
||||
one-off migration task that runs `python litellm/proxy/prisma_migration.py`
|
||||
against the backend image:
|
||||
one-off migration task that runs the `migrations` component of the same
|
||||
image:
|
||||
|
||||
- AWS: an `aws_ecs_task_definition` (`litellm-migrations`). Run with
|
||||
`aws ecs run-task` — the command is printed in `terraform output`.
|
||||
|
|
@ -234,19 +239,17 @@ dynamic-credentials OIDC).
|
|||
Required overrides the launcher must supply per stack:
|
||||
|
||||
- **AWS** (`terraform/litellm/aws`): `region`, `azs`, `tenant`, `env`.
|
||||
The image vars (`gateway_image`, `backend_image`, `ui_image`,
|
||||
`migrations_image`) can be left at their defaults — the GHCR images
|
||||
are anonymous-readable and ECS Fargate pulls them without extra
|
||||
credentials.
|
||||
`image` can be left at its default: the GHCR image is
|
||||
anonymous-readable and ECS Fargate pulls it without extra credentials.
|
||||
|
||||
- **GCP** (`terraform/litellm/gcp`): `project`, `tenant`, `env`, **and
|
||||
one of**:
|
||||
- `image_registry` pointed at an Artifact Registry **remote** repository
|
||||
backed by `https://ghcr.io` (e.g.
|
||||
`us-central1-docker.pkg.dev/<project>/litellm/berriai`), so Cloud Run
|
||||
pulls the four upstream `litellm-*` images through it; or
|
||||
- all four per-component `*_image` URIs pointing at images mirrored
|
||||
into a regular Artifact Registry repo.
|
||||
pulls the upstream `litellm` image through it; or
|
||||
- `image` pointing at a copy mirrored into a regular Artifact Registry
|
||||
repo.
|
||||
|
||||
The defaults (`ghcr.io/berriai`) cause Cloud Run admission to reject
|
||||
the service spec — Cloud Run only authenticates against Artifact
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ Deploys the componentized LiteLLM proxy on AWS:
|
|||
- LLM data-plane prefixes (`/v1/chat/*`, `/v1/embeddings`, …) → `gateway`
|
||||
- UI assets (`/`, `/_next/*`, `/litellm-asset-prefix/*`, …) → `ui`
|
||||
- Everything else (management API: `/key/*`, `/user/*`, …) → `backend`
|
||||
- **One-off migration task** (`litellm-migrations`) that runs `prisma migrate deploy` from the dedicated `ghcr.io/berriai/litellm-migrations` image
|
||||
- **One-off migration task** (`litellm-migrations`) that runs `prisma migrate deploy` through the `migrations` component of the same image
|
||||
|
||||
## Bring your own networking, database, and Redis
|
||||
|
||||
|
|
@ -248,8 +248,7 @@ this with `litellm_license`. To tune the export cadence, set
|
|||
(`python -m litellm.proxy.prometheus_metrics_server`) to the gateway task that
|
||||
aggregates the workers' samples over a shared task volume, so a scrape never
|
||||
runs on an inference worker. The ALB never routes to that port and the tasks
|
||||
security group only opens it to `gateway_metrics_scrape_cidrs`. Needs
|
||||
`gateway_image` v1.101.0 or newer. See
|
||||
security group only opens it to `gateway_metrics_scrape_cidrs`. See
|
||||
[Prometheus metrics](https://docs.litellm.ai/docs/proxy/prometheus) for the
|
||||
metrics themselves.
|
||||
|
||||
|
|
@ -285,10 +284,10 @@ tokens the workers used to (see [Aurora + IAM auth](#aurora--iam-auth)): the
|
|||
pooler mints a token from the task role, renews it before it expires and hands
|
||||
the workers a loopback URL with a static password instead
|
||||
|
||||
The componentized `gateway_image` starts through `python -m gateway.launch`,
|
||||
The image's `gateway` component starts through `python -m gateway.launch`,
|
||||
which reads these variables, starts the pooler once per task and hands the
|
||||
workers its loopback URL; the classic `litellm` image honours them the same
|
||||
way.
|
||||
workers its loopback URL; the monolithic `proxy` component honours them the
|
||||
same way.
|
||||
|
||||
### Scaling the gateway on requests and tokens
|
||||
|
||||
|
|
@ -499,12 +498,19 @@ top — set org-wide tags there, per-deployment tags via the `tags` input.
|
|||
|
||||
## Image pulls
|
||||
|
||||
The defaults pull from `ghcr.io/berriai/litellm-<component>:v1.86.0-dev`,
|
||||
which is anonymous-readable. There are four images: `litellm-gateway`,
|
||||
`litellm-backend`, `litellm-ui`, and `litellm-migrations` (slim image used
|
||||
only by the one-off migration task — runs `prisma migrate deploy` against
|
||||
the writer DB and exits). Bump them together when bumping LiteLLM. To pull
|
||||
from a private registry:
|
||||
`image` defaults to `ghcr.io/berriai/litellm:v1.104.0`, which is
|
||||
anonymous-readable. The gateway, backend and UI services, the metrics and
|
||||
collector sidecars and the one-off migration task all run that one image
|
||||
and pick their process through the entrypoint's first argument, so bumping
|
||||
LiteLLM is a single variable change. That entrypoint ships from v1.104.0;
|
||||
an older tag only knows how to run the monolithic proxy, so pinning one
|
||||
fails at container start.
|
||||
|
||||
Migrating from the retired `gateway_image`, `backend_image`, `ui_image`
|
||||
and `migrations_image` inputs: drop them and set `image` once. Terraform
|
||||
rejects the old names at plan time (`An argument named "gateway_image"
|
||||
is not expected here`), so a stale `.tfvars` cannot silently keep pulling
|
||||
the per-component images. To pull from a private registry:
|
||||
|
||||
- **ECR (same account)**: the execution role already has
|
||||
`AmazonECSTaskExecutionRolePolicy`, which grants ECR pull for repos in
|
||||
|
|
|
|||
|
|
@ -207,11 +207,18 @@ locals {
|
|||
|
||||
proxy_config_fetch_cmd = "python -c \"import os, boto3; boto3.client('s3', region_name=os.environ['AWS_REGION']).download_file(os.environ['LITELLM_PROXY_CONFIG_S3_BUCKET'], os.environ['LITELLM_PROXY_CONFIG_S3_KEY'], os.environ['CONFIG_FILE_PATH'])\""
|
||||
|
||||
# Gateway always needs --workers wired in (no NUM_WORKERS env var support
|
||||
# in the image entrypoint). When proxy_config is enabled we also have to
|
||||
# pull the config from S3 first, so the command goes through `sh -c`;
|
||||
# otherwise we keep the image's ENTRYPOINT and only override `command`.
|
||||
gateway_uvicorn_args = "--host 0.0.0.0 --port 4000 --workers ${var.gateway_num_workers}"
|
||||
image_entrypoint = "/app/docker-entrypoint.sh"
|
||||
component_args = {
|
||||
gateway = ["gateway", "--workers", tostring(var.gateway_num_workers)]
|
||||
backend = ["backend"]
|
||||
collector = ["collector"]
|
||||
}
|
||||
component_overrides = {
|
||||
for component, args in local.component_args : component => local.proxy_config_enabled ? {
|
||||
entryPoint = ["sh", "-c"]
|
||||
command = ["${local.proxy_config_fetch_cmd} && exec ${local.image_entrypoint} ${join(" ", args)}"]
|
||||
} : { command = args }
|
||||
}
|
||||
|
||||
gateway_pool_env = var.gateway_connection_pool_enabled ? [
|
||||
{ name = "LITELLM_PGBOUNCER_ENABLED", value = "true" },
|
||||
|
|
@ -228,11 +235,10 @@ locals {
|
|||
|
||||
gateway_metrics_container = local.metrics_enabled ? [
|
||||
{
|
||||
name = "metrics"
|
||||
image = var.gateway_image
|
||||
essential = false
|
||||
entryPoint = ["python", "-m", "litellm.proxy.prometheus_metrics_server"]
|
||||
command = ["--port", tostring(var.gateway_metrics_port)]
|
||||
name = "metrics"
|
||||
image = var.image
|
||||
essential = false
|
||||
command = ["metrics", "--port", tostring(var.gateway_metrics_port)]
|
||||
|
||||
portMappings = [{ containerPort = var.gateway_metrics_port, protocol = "tcp" }]
|
||||
environment = local.metrics_env
|
||||
|
|
@ -257,28 +263,6 @@ locals {
|
|||
}
|
||||
] : []
|
||||
|
||||
backend_uvicorn_args = "--host 0.0.0.0 --port 4001"
|
||||
|
||||
gateway_launch_cmd = "case \"$USE_DDTRACE\" in [Tt][Rr][Uu][Ee]) export DD_TRACE_OPENAI_ENABLED=\"False\"; exec ddtrace-run python -m gateway.launch ${local.gateway_uvicorn_args};; *) exec python -m gateway.launch ${local.gateway_uvicorn_args};; esac"
|
||||
backend_launch_cmd = "case \"$USE_DDTRACE\" in [Tt][Rr][Uu][Ee]) export DD_TRACE_OPENAI_ENABLED=\"False\"; exec ddtrace-run uvicorn backend.main:app ${local.backend_uvicorn_args};; *) exec uvicorn backend.main:app ${local.backend_uvicorn_args};; esac"
|
||||
|
||||
gateway_proxy_overrides = local.proxy_config_enabled ? {
|
||||
entryPoint = ["sh", "-c"]
|
||||
command = [
|
||||
"${local.proxy_config_fetch_cmd} && ${local.gateway_launch_cmd}"
|
||||
]
|
||||
} : {
|
||||
entryPoint = ["sh", "-c"]
|
||||
command = [local.gateway_launch_cmd]
|
||||
}
|
||||
|
||||
backend_proxy_overrides = local.proxy_config_enabled ? {
|
||||
entryPoint = ["sh", "-c"]
|
||||
command = [
|
||||
"${local.proxy_config_fetch_cmd} && ${local.backend_launch_cmd}"
|
||||
]
|
||||
} : {}
|
||||
|
||||
collector_address = "tcp://127.0.0.1:${var.collector_port}"
|
||||
collector_env = var.collector_enabled ? [
|
||||
{ name = "LITELLM_COLLECTOR_ENABLED", value = "true" },
|
||||
|
|
@ -299,22 +283,15 @@ locals {
|
|||
local.collector_env,
|
||||
)
|
||||
|
||||
collector_launch_cmd = "exec python -m litellm.proxy.collector"
|
||||
collector_command = [
|
||||
local.proxy_config_enabled ? "${local.proxy_config_fetch_cmd} && ${local.collector_launch_cmd}" : local.collector_launch_cmd
|
||||
]
|
||||
|
||||
collector_container = var.collector_enabled ? [{
|
||||
collector_container = var.collector_enabled ? [merge({
|
||||
name = "collector"
|
||||
image = var.gateway_image
|
||||
image = var.image
|
||||
essential = false
|
||||
cpu = var.collector_cpu
|
||||
memory = var.collector_memory
|
||||
|
||||
restartPolicy = { enabled = true }
|
||||
|
||||
entryPoint = ["sh", "-c"]
|
||||
command = local.collector_command
|
||||
environment = concat(
|
||||
local.shared_env,
|
||||
local.gateway_extra_env_list,
|
||||
|
|
@ -333,7 +310,7 @@ locals {
|
|||
awslogs-stream-prefix = "collector"
|
||||
}
|
||||
}
|
||||
}] : []
|
||||
}, local.component_overrides.collector)] : []
|
||||
}
|
||||
|
||||
# ---------- Gateway ----------
|
||||
|
|
@ -389,7 +366,7 @@ resource "aws_ecs_task_definition" "gateway" {
|
|||
merge(
|
||||
{
|
||||
name = "gateway"
|
||||
image = var.gateway_image
|
||||
image = var.image
|
||||
essential = true
|
||||
|
||||
portMappings = [{ containerPort = 4000, protocol = "tcp" }]
|
||||
|
|
@ -410,7 +387,7 @@ resource "aws_ecs_task_definition" "gateway" {
|
|||
}
|
||||
}
|
||||
},
|
||||
local.gateway_proxy_overrides,
|
||||
local.component_overrides.gateway,
|
||||
)
|
||||
], local.gateway_metrics_container, local.collector_container))
|
||||
|
||||
|
|
@ -499,7 +476,7 @@ resource "aws_ecs_task_definition" "backend" {
|
|||
merge(
|
||||
{
|
||||
name = "backend"
|
||||
image = var.backend_image
|
||||
image = var.image
|
||||
essential = true
|
||||
|
||||
portMappings = [{ containerPort = 4001, protocol = "tcp" }]
|
||||
|
|
@ -522,7 +499,7 @@ resource "aws_ecs_task_definition" "backend" {
|
|||
}
|
||||
}
|
||||
},
|
||||
local.backend_proxy_overrides,
|
||||
local.component_overrides.backend,
|
||||
)
|
||||
])
|
||||
|
||||
|
|
@ -591,8 +568,9 @@ resource "aws_ecs_task_definition" "ui" {
|
|||
container_definitions = jsonencode([
|
||||
{
|
||||
name = "ui"
|
||||
image = var.ui_image
|
||||
image = var.image
|
||||
essential = true
|
||||
command = ["ui"]
|
||||
portMappings = [{ containerPort = 3000, protocol = "tcp" }]
|
||||
|
||||
logConfiguration = {
|
||||
|
|
|
|||
|
|
@ -1,16 +1,8 @@
|
|||
# Task definition for the dedicated litellm-migrations image. Mirrors the
|
||||
# pre-install/pre-upgrade Helm hook in helm/litellm/templates/migrations-job.yaml.
|
||||
#
|
||||
# The image (built from migrations/Dockerfile) ships with
|
||||
# `ENTRYPOINT ["python3", "/app/run.py"]`. run.py assembles DATABASE_URL from
|
||||
# the discrete DATABASE_* env vars (IAM auth here) via DatabaseURLSettings,
|
||||
# then calls ProxyExtrasDBManager.setup_database() — i.e. `prisma migrate
|
||||
# deploy` with the v2 resolver and P3005/P3009/P3018 recovery. It does NOT
|
||||
# read CONFIG_FILE_PATH, the master key, or DISABLE_SCHEMA_UPDATE, so we
|
||||
# don't pass them.
|
||||
#
|
||||
# Invoked automatically by `terraform_data.migration` in bootstrap.tf during
|
||||
# every apply (after the IAM-authed user has been created). The
|
||||
# Runs the `migrations` component of var.image: run.py assembles DATABASE_URL
|
||||
# from the discrete DATABASE_* env vars (IAM auth here) and runs `prisma
|
||||
# migrate deploy`. It does not read CONFIG_FILE_PATH, the master key, or
|
||||
# DISABLE_SCHEMA_UPDATE, so we don't pass them. Invoked automatically by
|
||||
# `terraform_data.migration` in bootstrap.tf during every apply; the
|
||||
# `migration_run_command` output is preserved for break-glass manual re-runs.
|
||||
resource "aws_ecs_task_definition" "migrations" {
|
||||
count = local.database_enabled ? 1 : 0
|
||||
|
|
@ -28,10 +20,10 @@ resource "aws_ecs_task_definition" "migrations" {
|
|||
|
||||
container_definitions = jsonencode([{
|
||||
name = "migrations"
|
||||
image = var.migrations_image
|
||||
image = var.image
|
||||
essential = true
|
||||
command = ["migrations"]
|
||||
|
||||
# No entryPoint/command override — the image's ENTRYPOINT runs run.py.
|
||||
environment = local.shared_env
|
||||
secrets = local.byo_database_secrets
|
||||
|
||||
|
|
|
|||
|
|
@ -66,14 +66,14 @@ run "enabled_adds_a_sidecar_that_shares_the_gateway_transport" {
|
|||
|
||||
assert {
|
||||
condition = (
|
||||
local.collector_container[0].image == var.gateway_image &&
|
||||
local.collector_container[0].entryPoint == ["sh", "-c"] &&
|
||||
local.collector_container[0].command == ["exec python -m litellm.proxy.collector"] &&
|
||||
local.collector_container[0].image == var.image &&
|
||||
!can(local.collector_container[0].entryPoint) &&
|
||||
local.collector_container[0].command == tolist(["collector"]) &&
|
||||
local.collector_container[0].essential == false &&
|
||||
local.collector_container[0].restartPolicy.enabled == true &&
|
||||
{ for e in local.collector_container[0].environment : e.name => e.value }["LITELLM_JOB_ROLE"] == "collector"
|
||||
)
|
||||
error_message = "The sidecar must run litellm.proxy.collector from the gateway image as a restartable, non-essential collector."
|
||||
error_message = "The sidecar must run the collector component of the shared image as a restartable, non-essential collector."
|
||||
}
|
||||
|
||||
assert {
|
||||
|
|
@ -108,8 +108,9 @@ run "proxy_config_is_fetched_by_the_sidecar_too" {
|
|||
|
||||
assert {
|
||||
condition = (
|
||||
local.collector_container[0].entryPoint == tolist(["sh", "-c"]) &&
|
||||
startswith(local.collector_container[0].command[0], local.proxy_config_fetch_cmd) &&
|
||||
endswith(local.collector_container[0].command[0], "exec python -m litellm.proxy.collector") &&
|
||||
endswith(local.collector_container[0].command[0], "exec /app/docker-entrypoint.sh collector") &&
|
||||
contains([for e in local.collector_container[0].environment : e.name], "CONFIG_FILE_PATH")
|
||||
)
|
||||
error_message = "The sidecar must pull the proxy config from S3 before starting, like the gateway does."
|
||||
|
|
|
|||
|
|
@ -116,12 +116,10 @@ run "gateway_starts_through_the_pool_aware_launcher" {
|
|||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
strcontains(local.gateway_launch_cmd, "exec python -m gateway.launch --host 0.0.0.0 --port 4000 --workers 4"),
|
||||
strcontains(local.gateway_launch_cmd, "exec ddtrace-run python -m gateway.launch --host 0.0.0.0 --port 4000 --workers 4"),
|
||||
!strcontains(local.gateway_launch_cmd, "uvicorn gateway.main:app"),
|
||||
local.gateway_proxy_overrides.command[0] == local.gateway_launch_cmd,
|
||||
local.component_overrides.gateway.command == tolist(["gateway", "--workers", "4"]),
|
||||
!can(local.component_overrides.gateway.entryPoint),
|
||||
])
|
||||
error_message = "The gateway must start through gateway.launch (with and without ddtrace) so the pooler starts once before uvicorn forks the workers."
|
||||
error_message = "The gateway must start through the image's gateway component with the configured worker count so the pooler starts once before uvicorn forks the workers."
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -55,14 +55,15 @@ run "metrics_port_adds_a_sidecar_volume_and_scrape_rule" {
|
|||
length(local.gateway_metrics_container) == 1,
|
||||
local.gateway_metrics_container[0].name == "metrics",
|
||||
local.gateway_metrics_container[0].essential == false,
|
||||
join(" ", local.gateway_metrics_container[0].entryPoint) == "python -m litellm.proxy.prometheus_metrics_server",
|
||||
join(" ", local.gateway_metrics_container[0].command) == "--port 9464",
|
||||
local.gateway_metrics_container[0].image == var.image,
|
||||
!can(local.gateway_metrics_container[0].entryPoint),
|
||||
join(" ", local.gateway_metrics_container[0].command) == "metrics --port 9464",
|
||||
one(local.gateway_metrics_container[0].portMappings).containerPort == 9464,
|
||||
one(local.gateway_metrics_container[0].environment).value == "/tmp/litellm_prometheus_multiproc",
|
||||
one(local.gateway_metrics_container[0].mountPoints).sourceVolume == "prometheus-multiproc",
|
||||
strcontains(local.gateway_metrics_container[0].healthCheck.command[3], "9464"),
|
||||
])
|
||||
error_message = "The metrics sidecar must run prometheus_metrics_server on the configured port, share the multiproc volume, and health-check that port."
|
||||
error_message = "The metrics sidecar must run the shared image's metrics component on the configured port, share the multiproc volume, and health-check that port."
|
||||
}
|
||||
|
||||
assert {
|
||||
|
|
|
|||
79
terraform/litellm/aws/tests/unified_image.tftest.hcl
Normal file
79
terraform/litellm/aws/tests/unified_image.tftest.hcl
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
# Plan-only coverage for the single-image contract: every task pulls var.image
|
||||
# and names its process through the image entrypoint's component argument.
|
||||
# Gateway, backend and migrations container_definitions are unknown at plan
|
||||
# time (Aurora/ElastiCache endpoints), so those assertions target the locals
|
||||
# they are built from; the UI task is fully known and is checked as rendered.
|
||||
|
||||
mock_provider "aws" {
|
||||
mock_data "aws_iam_policy_document" {
|
||||
defaults = {
|
||||
json = "{\"Version\":\"2012-10-17\",\"Statement\":[]}"
|
||||
}
|
||||
}
|
||||
}
|
||||
mock_provider "random" {}
|
||||
|
||||
variables {
|
||||
region = "us-east-1"
|
||||
tenant = "acme"
|
||||
env = "test"
|
||||
allow_plaintext_alb = true
|
||||
azs = ["us-east-1a", "us-east-1b"]
|
||||
image = "123456789012.dkr.ecr.us-east-1.amazonaws.com/litellm:test"
|
||||
gateway_num_workers = 3
|
||||
gateway_metrics_port = 9464
|
||||
collector_enabled = true
|
||||
}
|
||||
|
||||
run "every_container_runs_a_component_of_the_one_image" {
|
||||
command = plan
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
jsondecode(aws_ecs_task_definition.ui.container_definitions)[0].image == var.image,
|
||||
join(" ", jsondecode(aws_ecs_task_definition.ui.container_definitions)[0].command) == "ui",
|
||||
!can(jsondecode(aws_ecs_task_definition.ui.container_definitions)[0].entryPoint),
|
||||
])
|
||||
error_message = "The UI task must run the ui component of var.image through the image entrypoint."
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
join(" ", local.component_overrides.gateway.command) == "gateway --workers 3",
|
||||
join(" ", local.component_overrides.backend.command) == "backend",
|
||||
join(" ", local.component_overrides.collector.command) == "collector",
|
||||
alltrue([for component in keys(local.component_args) : !can(local.component_overrides[component].entryPoint)]),
|
||||
local.gateway_metrics_container[0].image == var.image,
|
||||
join(" ", local.gateway_metrics_container[0].command) == "metrics --port 9464",
|
||||
!can(local.gateway_metrics_container[0].entryPoint),
|
||||
local.collector_container[0].image == var.image,
|
||||
])
|
||||
error_message = "Gateway, backend and the sidecars must keep the image entrypoint and select their component (plus the gateway worker count) through command."
|
||||
}
|
||||
}
|
||||
|
||||
run "proxy_config_wraps_the_fetch_around_the_same_component" {
|
||||
command = plan
|
||||
|
||||
variables {
|
||||
proxy_config = { model_list = [] }
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
for component, args in local.component_args :
|
||||
join(" ", local.component_overrides[component].entryPoint) == "sh -c" &&
|
||||
startswith(local.component_overrides[component].command[0], local.proxy_config_fetch_cmd) &&
|
||||
endswith(local.component_overrides[component].command[0], " && exec /app/docker-entrypoint.sh ${join(" ", args)}")
|
||||
])
|
||||
error_message = "With proxy_config, gateway, backend and collector must download the config and then exec the image entrypoint with their component."
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
!can(local.gateway_metrics_container[0].entryPoint),
|
||||
!can(jsondecode(aws_ecs_task_definition.ui.container_definitions)[0].entryPoint),
|
||||
])
|
||||
error_message = "The metrics sidecar and the UI never read the proxy config and must keep the plain image entrypoint."
|
||||
}
|
||||
}
|
||||
|
|
@ -135,38 +135,17 @@ variable "azs" {
|
|||
|
||||
# ---------- Component images ----------
|
||||
#
|
||||
# Defaults pin the four componentized images at the same release tag on
|
||||
# GHCR. Override on a per-component basis in tfvars when bumping; bump them
|
||||
# together when bumping the LiteLLM release.
|
||||
|
||||
variable "gateway_image" {
|
||||
description = "Container image for the gateway (data plane, port 4000). Tag must match a tag actually published to GHCR — the split images use the `v`-prefixed semver convention."
|
||||
type = string
|
||||
default = "ghcr.io/berriai/litellm-gateway:v1.86.0-dev"
|
||||
}
|
||||
|
||||
variable "backend_image" {
|
||||
description = "Container image for the backend (management API, port 4001)."
|
||||
type = string
|
||||
default = "ghcr.io/berriai/litellm-backend:v1.86.0-dev"
|
||||
}
|
||||
|
||||
variable "ui_image" {
|
||||
description = "Container image for the UI (nginx static export, port 3000)."
|
||||
type = string
|
||||
default = "ghcr.io/berriai/litellm-ui:v1.86.0-dev"
|
||||
}
|
||||
|
||||
variable "migrations_image" {
|
||||
variable "image" {
|
||||
description = <<-EOT
|
||||
Container image for the one-off prisma migration task. Built from
|
||||
`migrations/Dockerfile` — slim image whose ENTRYPOINT runs
|
||||
`python3 /app/run.py` (assembles DATABASE_URL from DATABASE_* env vars
|
||||
via DatabaseURLSettings, then runs `prisma migrate deploy`). Should track
|
||||
the same release tag as gateway/backend/ui.
|
||||
The one LiteLLM image every task runs. Its entrypoint picks the process
|
||||
from the first argument (`gateway`, `backend`, `ui`, `migrations`,
|
||||
`metrics`, `collector`), so the gateway, backend, UI and migration task
|
||||
definitions and the sidecars all pull this URI. The component entrypoint
|
||||
ships from v1.104.0; older tags only run the monolithic proxy. Tag must
|
||||
match a tag actually published to GHCR (`v<semver>` for releases).
|
||||
EOT
|
||||
type = string
|
||||
default = "ghcr.io/berriai/litellm-migrations:v1.86.0-dev"
|
||||
default = "ghcr.io/berriai/litellm:v1.104.0"
|
||||
}
|
||||
|
||||
# ---------- Service sizing ----------
|
||||
|
|
@ -634,13 +613,11 @@ variable "gateway_metrics_port" {
|
|||
description = <<-EOT
|
||||
Serve Prometheus /metrics from a `metrics` sidecar container in the
|
||||
gateway task on this port (1-65535, not 4000), so a scrape never runs on
|
||||
an inference worker. The sidecar runs the gateway image with
|
||||
`python -m litellm.proxy.prometheus_metrics_server` and aggregates the
|
||||
workers' PROMETHEUS_MULTIPROC_DIR samples over a task volume. Null (the
|
||||
default) leaves /metrics on the gateway port only. The sidecar port has
|
||||
no virtual-key auth and is not routed through the ALB; open it to your
|
||||
scrapers with gateway_metrics_scrape_cidrs. Needs gateway_image v1.101.0
|
||||
or newer.
|
||||
an inference worker. The sidecar runs the `metrics` component of `image`
|
||||
and aggregates the workers' PROMETHEUS_MULTIPROC_DIR samples over a task
|
||||
volume. Null (the default) leaves /metrics on the gateway port only. The
|
||||
sidecar port has no virtual-key auth and is not routed through the ALB;
|
||||
open it to your scrapers with gateway_metrics_scrape_cidrs.
|
||||
EOT
|
||||
type = number
|
||||
default = null
|
||||
|
|
@ -812,7 +789,7 @@ variable "billing_metrics_ca_cert_pem" {
|
|||
# ---------- Collector sidecar ----------
|
||||
#
|
||||
# Opt-in offload of spend tracking from the gateway's uvicorn workers to a
|
||||
# `python -m litellm.proxy.collector` sidecar in the same Fargate task (helm's
|
||||
# `collector` component sidecar in the same Fargate task (helm's
|
||||
# `gateway.collector`). Fargate awsvpc tasks share one network namespace,
|
||||
# so the sidecar listens on loopback TCP. Disabled (the default) adds nothing
|
||||
# to the task definition.
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ Deploys the componentized LiteLLM proxy on GCP:
|
|||
- **Secret Manager** entries for `LITELLM_MASTER_KEY` and `DATABASE_PASSWORD`
|
||||
- **Cloud Run v2** services for `gateway` (port 4000), `backend` (port 4001),
|
||||
and `ui` (port 3000), all using a shared runtime service account
|
||||
- **Cloud Run Job** (`litellm-migrations`) that runs `prisma migrate deploy` from the dedicated `ghcr.io/berriai/litellm-migrations` image
|
||||
- **Cloud Run Job** (`litellm-migrations`) that runs `prisma migrate deploy` through the `migrations` component of the same image
|
||||
- **External global HTTP(S) load balancer** with serverless NEGs and a URL
|
||||
map mirroring the helm-chart ingress path routing:
|
||||
- LLM data-plane prefixes → `gateway`
|
||||
|
|
@ -24,18 +24,18 @@ Deploys the componentized LiteLLM proxy on GCP:
|
|||
|
||||
## Image pulls
|
||||
|
||||
There are four images: `litellm-gateway`, `litellm-backend`, `litellm-ui`,
|
||||
and `litellm-migrations` (slim image used only by the one-off Cloud Run
|
||||
Job — runs `prisma migrate deploy` against the writer DB and exits).
|
||||
Bump them together when bumping LiteLLM.
|
||||
Every Cloud Run service, sidecar and the migrations Job runs the one
|
||||
`litellm` image and picks its process through the entrypoint's first
|
||||
argument (`gateway`, `backend`, `ui`, `migrations`, `metrics`,
|
||||
`collector`), so bumping LiteLLM is a single `image_tag` change.
|
||||
|
||||
**Required override.** The `image_registry` default (`ghcr.io/berriai`)
|
||||
does **not** work as-is — Cloud Run only accepts images from Artifact
|
||||
Registry, `[region.]gcr.io`, or `docker.io`, and rejects `ghcr.io` URIs
|
||||
at apply time. Every deploy (including HCP Terraform 1-click) must
|
||||
supply either `image_registry` pointed at an Artifact Registry remote
|
||||
repo backed by GHCR, or full per-component `*_image` URIs against
|
||||
images you've already mirrored. The default is present only so
|
||||
repo backed by GHCR, or a full `image` URI for a copy you've already
|
||||
mirrored. The default is present only so
|
||||
`terraform plan` succeeds during local iteration.
|
||||
|
||||
**One-time setup (per project):** create a remote repo and let Cloud Run
|
||||
|
|
@ -54,13 +54,21 @@ Then point the stack at it via `image_registry`:
|
|||
|
||||
```hcl
|
||||
image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
|
||||
image_tag = "v1.86.0-dev"
|
||||
image_tag = "v1.104.0"
|
||||
```
|
||||
|
||||
The four `litellm-<component>:${image_tag}` URIs are composed from those
|
||||
two vars. Set `gateway_image` / `backend_image` / `ui_image` /
|
||||
`migrations_image` only if you need a per-component override (custom
|
||||
build, different tag).
|
||||
The `<image_registry>/litellm:<image_tag>` URI is composed from those two
|
||||
vars. Set `image` only if you need a full override (custom build,
|
||||
digest pin).
|
||||
|
||||
The component entrypoint ships from v1.104.0; an older tag only knows how
|
||||
to run the monolithic proxy, so pinning one fails at container start.
|
||||
|
||||
Migrating from the retired `gateway_image`, `backend_image`, `ui_image`
|
||||
and `migrations_image` inputs: drop them and set `image_registry` +
|
||||
`image_tag` (or `image`) once. Terraform rejects the old names at plan
|
||||
time, so a stale `.tfvars` cannot silently keep pulling the per-component
|
||||
images.
|
||||
|
||||
Two further notes:
|
||||
|
||||
|
|
@ -254,8 +262,7 @@ stack also adds Google's
|
|||
Secret Manager that scrapes `localhost:<port>/metrics` every 30s and writes to
|
||||
Cloud Monitoring as `prometheus.googleapis.com/...` metrics. Enabling it grants
|
||||
the runtime service account `roles/monitoring.metricWriter` and
|
||||
`roles/logging.logWriter` on the project. Needs `gateway_image` v1.101.0 or
|
||||
newer. See [Prometheus metrics](https://docs.litellm.ai/docs/proxy/prometheus)
|
||||
`roles/logging.logWriter` on the project. See [Prometheus metrics](https://docs.litellm.ai/docs/proxy/prometheus)
|
||||
for the metrics themselves
|
||||
|
||||
```hcl
|
||||
|
|
|
|||
|
|
@ -7,8 +7,8 @@
|
|||
# so they don't go live until the schema is in place.
|
||||
#
|
||||
# Triggers:
|
||||
# - re-runs if the migrations image changes (new release ships new prisma
|
||||
# migration files).
|
||||
# - re-runs if the image changes (new release ships new prisma migration
|
||||
# files).
|
||||
# - re-runs if the migration job is recreated.
|
||||
#
|
||||
# Requires `gcloud` on the machine running terraform, with user creds live
|
||||
|
|
@ -19,7 +19,7 @@ resource "terraform_data" "migration" {
|
|||
|
||||
triggers_replace = {
|
||||
job_id = google_cloud_run_v2_job.migrations[0].id
|
||||
job_image = local.migrations_image
|
||||
job_image = local.image
|
||||
}
|
||||
|
||||
provisioner "local-exec" {
|
||||
|
|
|
|||
|
|
@ -144,16 +144,12 @@ locals {
|
|||
{ name = "LITELLM_PGBOUNCER_MAX_CLIENT_CONN", value = tostring(var.gateway_pool_max_client_conn) },
|
||||
] : []
|
||||
|
||||
gateway_uvicorn_args = "--host 0.0.0.0 --port 4000 --workers ${var.gateway_num_workers}"
|
||||
backend_uvicorn_args = "--host 0.0.0.0 --port 4001"
|
||||
|
||||
gateway_launch_cmd = "case \"$USE_DDTRACE\" in [Tt][Rr][Uu][Ee]) export DD_TRACE_OPENAI_ENABLED=\"False\"; exec ddtrace-run python -m gateway.launch ${local.gateway_uvicorn_args};; *) exec python -m gateway.launch ${local.gateway_uvicorn_args};; esac"
|
||||
backend_launch_cmd = "case \"$USE_DDTRACE\" in [Tt][Rr][Uu][Ee]) export DD_TRACE_OPENAI_ENABLED=\"False\"; exec ddtrace-run uvicorn backend.main:app ${local.backend_uvicorn_args};; *) exec uvicorn backend.main:app ${local.backend_uvicorn_args};; esac"
|
||||
image_entrypoint = "/app/docker-entrypoint.sh"
|
||||
|
||||
gateway_args = join(" && ", concat(
|
||||
local.redis_ca_fragment,
|
||||
local.database_url_fragment,
|
||||
[local.gateway_launch_cmd],
|
||||
["exec ${local.image_entrypoint} gateway --workers ${var.gateway_num_workers}"],
|
||||
))
|
||||
|
||||
metrics_enabled = var.create_runtime && var.gateway_metrics_port != null
|
||||
|
|
@ -174,7 +170,7 @@ locals {
|
|||
backend_args = join(" && ", concat(
|
||||
local.redis_ca_fragment,
|
||||
local.database_url_fragment,
|
||||
[local.backend_launch_cmd],
|
||||
["exec ${local.image_entrypoint} backend"],
|
||||
))
|
||||
|
||||
collector_address = "tcp://127.0.0.1:${var.collector_port}"
|
||||
|
|
@ -202,13 +198,12 @@ locals {
|
|||
collector_args = join(" && ", concat(
|
||||
local.redis_ca_fragment,
|
||||
local.database_url_fragment,
|
||||
["exec python -m litellm.proxy.collector"],
|
||||
["exec ${local.image_entrypoint} collector"],
|
||||
))
|
||||
|
||||
# Env shipped to the migrations Job. The migrations image runs run.py
|
||||
# which assembles DATABASE_URL from these discrete vars itself, so we
|
||||
# only need writer-side DB env (no read replica, no proxy_config, no
|
||||
# master key).
|
||||
# run.py assembles DATABASE_URL from these discrete vars itself, so the
|
||||
# migrations Job only needs writer-side DB env (no read replica, no
|
||||
# proxy_config, no master key).
|
||||
migrations_env_kv = [
|
||||
{ name = "DATABASE_HOST", value = google_sql_database_instance.writer.private_ip_address },
|
||||
{ name = "DATABASE_PORT", value = "5432" },
|
||||
|
|
@ -254,7 +249,7 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
|
||||
containers {
|
||||
name = "gateway"
|
||||
image = local.gateway_image
|
||||
image = local.image
|
||||
command = ["sh", "-c"]
|
||||
args = [local.gateway_args]
|
||||
|
||||
|
|
@ -330,10 +325,9 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
dynamic "containers" {
|
||||
for_each = local.metrics_enabled ? [1] : []
|
||||
content {
|
||||
name = "metrics"
|
||||
image = local.gateway_image
|
||||
command = ["python", "-m", "litellm.proxy.prometheus_metrics_server"]
|
||||
args = ["--port", tostring(var.gateway_metrics_port)]
|
||||
name = "metrics"
|
||||
image = local.image
|
||||
args = ["metrics", "--port", tostring(var.gateway_metrics_port)]
|
||||
|
||||
dynamic "env" {
|
||||
for_each = local.metrics_env_kv
|
||||
|
|
@ -396,7 +390,7 @@ resource "google_cloud_run_v2_service" "gateway" {
|
|||
for_each = var.collector_enabled ? [1] : []
|
||||
content {
|
||||
name = "spend-collector"
|
||||
image = local.gateway_image
|
||||
image = local.image
|
||||
command = ["sh", "-c"]
|
||||
args = [local.collector_args]
|
||||
|
||||
|
|
@ -519,7 +513,7 @@ resource "google_cloud_run_v2_service" "backend" {
|
|||
}
|
||||
|
||||
containers {
|
||||
image = local.backend_image
|
||||
image = local.image
|
||||
command = ["sh", "-c"]
|
||||
args = [local.backend_args]
|
||||
|
||||
|
|
@ -635,7 +629,8 @@ resource "google_cloud_run_v2_service" "ui" {
|
|||
}
|
||||
|
||||
containers {
|
||||
image = local.ui_image
|
||||
image = local.image
|
||||
args = ["ui"]
|
||||
|
||||
ports {
|
||||
container_port = 3000
|
||||
|
|
@ -697,9 +692,6 @@ resource "google_cloud_run_v2_service_iam_member" "ui_allusers" {
|
|||
}
|
||||
|
||||
# ---------- Migrations job ----------
|
||||
# Dedicated litellm-migrations image — slim, ENTRYPOINT runs run.py which
|
||||
# assembles DATABASE_URL from the DATABASE_* env vars and runs `prisma
|
||||
# migrate deploy`. No proxy_config, no master key, no shell wrapper.
|
||||
resource "google_cloud_run_v2_job" "migrations" {
|
||||
count = var.create_runtime ? 1 : 0
|
||||
|
||||
|
|
@ -718,7 +710,8 @@ resource "google_cloud_run_v2_job" "migrations" {
|
|||
}
|
||||
|
||||
containers {
|
||||
image = local.migrations_image
|
||||
image = local.image
|
||||
args = ["migrations"]
|
||||
|
||||
# Prisma's Node + Rust engine plus the v2 migration resolver
|
||||
# routinely peaks above 1 GiB while applying the schema, so 2 GiB
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ gcloud services enable \
|
|||
|
||||
## Create the Artifact Registry passthrough to GHCR
|
||||
|
||||
Cloud Run only pulls from Artifact Registry, `gcr.io`, or `docker.io`; it rejects `ghcr.io` URIs at apply time. The four LiteLLM images live on GHCR, so the stack needs a remote Artifact Registry repo pointed at GHCR. This is a one-time setup per project.
|
||||
Cloud Run only pulls from Artifact Registry, `gcr.io`, or `docker.io`; it rejects `ghcr.io` URIs at apply time. The LiteLLM image lives on GHCR, so the stack needs a remote Artifact Registry repo pointed at GHCR. This is a one-time setup per project.
|
||||
|
||||
```bash
|
||||
gcloud artifacts repositories create litellm \
|
||||
|
|
|
|||
|
|
@ -24,12 +24,12 @@
|
|||
},
|
||||
{
|
||||
"name": "image_tag",
|
||||
"description": "Tag for the four litellm-* images (gateway, backend, ui, migrations). Bump together when bumping LiteLLM",
|
||||
"default": "v1.86.0-dev"
|
||||
"description": "Tag of the litellm image every workload (gateway, backend, ui, migrations) runs",
|
||||
"default": "v1.104.0"
|
||||
},
|
||||
{
|
||||
"name": "image_registry",
|
||||
"description": "Artifact Registry path prefix for the four litellm-* images. Format: <region>-docker.pkg.dev/<project>/litellm/berriai, pointing at the remote repo you created above. Substitute BOTH REGION and PROJECT_ID in the default to match the AR repo you just created (REGION must match the region you picked above). The ghcr.io/berriai default in the module does NOT work; Cloud Run rejects ghcr.io URIs at apply time",
|
||||
"description": "Artifact Registry path prefix for the litellm image. Format: <region>-docker.pkg.dev/<project>/litellm/berriai, pointing at the remote repo you created above. Substitute BOTH REGION and PROJECT_ID in the default to match the AR repo you just created (REGION must match the region you picked above). The ghcr.io/berriai default in the module does NOT work; Cloud Run rejects ghcr.io URIs at apply time",
|
||||
"default": "REGION-docker.pkg.dev/PROJECT_ID/litellm/berriai"
|
||||
},
|
||||
{
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@
|
|||
# "Using as a module" section.
|
||||
#
|
||||
# Knobs not surfaced as variables here (per-component sizing/instances,
|
||||
# Cloud SQL tier/edition, Memorystore tier, per-component image overrides)
|
||||
# Cloud SQL tier/edition, Memorystore tier, a full `image` override)
|
||||
# can be set directly on this block — see ../../variables.tf.
|
||||
module "litellm" {
|
||||
source = "../../"
|
||||
|
|
|
|||
|
|
@ -38,12 +38,12 @@ env = "stage"
|
|||
|
||||
# Images. Cloud Run rejects ghcr.io, so a real deploy must point
|
||||
# image_registry at an Artifact Registry remote repo (see README "Image
|
||||
# pulls"); image_tag is applied to all four litellm-* images. Per-component
|
||||
# *_image overrides are NOT exposed here — set them directly on the
|
||||
# pulls"); every workload runs <image_registry>/litellm:<image_tag>. The
|
||||
# full `image` override is NOT exposed here: set it directly on the
|
||||
# `module "litellm"` block in main.tf (see ../../variables.tf) if you need
|
||||
# to mix-and-match versions.
|
||||
# a custom build.
|
||||
# image_registry = "us-central1-docker.pkg.dev/my-gcp-project/litellm/berriai"
|
||||
# image_tag = "v1.86.0-dev"
|
||||
# image_tag = "v1.104.0"
|
||||
|
||||
# ---------- proxy_config (mirrors helm gateway.config.proxy_config) ----------
|
||||
# proxy_config = {
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
# Curated surface for the one-command deploy path. The module (../../)
|
||||
# exposes far more knobs (per-component CPU/memory/instances, Cloud SQL
|
||||
# tier/edition, Memorystore tier, per-component image overrides, …). To
|
||||
# tier/edition, Memorystore tier, a full `image` override, …). To
|
||||
# tune those, set them directly on the `module "litellm"` block in
|
||||
# main.tf, or call the module from your own root config. Full per-variable
|
||||
# docs live in ../../variables.tf — the module is the source of truth.
|
||||
|
|
@ -75,17 +75,17 @@ variable "ui_password" {
|
|||
|
||||
# Image source. Cloud Run rejects ghcr.io, so a real deploy must point
|
||||
# image_registry at an Artifact Registry remote repo (see README "Image
|
||||
# pulls"). Per-component overrides live in ../../variables.tf.
|
||||
# pulls"). The full `image` override lives in ../../variables.tf.
|
||||
variable "image_registry" {
|
||||
description = "Registry path prefix; images composed as <image_registry>/litellm-<component>:<image_tag>."
|
||||
description = "Registry path prefix; the image is composed as <image_registry>/litellm:<image_tag>."
|
||||
type = string
|
||||
default = "ghcr.io/berriai"
|
||||
}
|
||||
|
||||
variable "image_tag" {
|
||||
description = "Tag applied to all four litellm-* images. Bump in lockstep."
|
||||
description = "Tag of the litellm image every workload runs."
|
||||
type = string
|
||||
default = "v1.86.0-dev"
|
||||
default = "v1.104.0"
|
||||
}
|
||||
|
||||
# TLS — provide DNS names for a managed cert, or opt into HTTP-only for dev.
|
||||
|
|
|
|||
|
|
@ -92,11 +92,5 @@ locals {
|
|||
{ name = "PROXY_CONFIG_HASH", value = md5(local.proxy_config_yaml) },
|
||||
] : []
|
||||
|
||||
# Resolved image URIs: per-component override wins, otherwise compose
|
||||
# from image_registry + image_tag. Cloud Run only accepts AR / gcr.io /
|
||||
# docker.io paths — see variables.tf for the full constraint list.
|
||||
gateway_image = var.gateway_image != "" ? var.gateway_image : "${var.image_registry}/litellm-gateway:${var.image_tag}"
|
||||
backend_image = var.backend_image != "" ? var.backend_image : "${var.image_registry}/litellm-backend:${var.image_tag}"
|
||||
ui_image = var.ui_image != "" ? var.ui_image : "${var.image_registry}/litellm-ui:${var.image_tag}"
|
||||
migrations_image = var.migrations_image != "" ? var.migrations_image : "${var.image_registry}/litellm-migrations:${var.image_tag}"
|
||||
image = var.image != "" ? var.image : "${var.image_registry}/litellm:${var.image_tag}"
|
||||
}
|
||||
|
|
|
|||
|
|
@ -64,14 +64,14 @@ run "enabled_adds_a_sidecar_that_shares_the_gateway_transport" {
|
|||
|
||||
assert {
|
||||
condition = (
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].image == local.gateway_image &&
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].image == google_cloud_run_v2_service.gateway[0].template[0].containers[0].image &&
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].command == tolist(["sh", "-c"]) &&
|
||||
endswith(google_cloud_run_v2_service.gateway[0].template[0].containers[1].args[0], " && exec python -m litellm.proxy.collector") &&
|
||||
endswith(google_cloud_run_v2_service.gateway[0].template[0].containers[1].args[0], " && exec /app/docker-entrypoint.sh collector") &&
|
||||
strcontains(google_cloud_run_v2_service.gateway[0].template[0].containers[1].args[0], "export DATABASE_URL=") &&
|
||||
strcontains(google_cloud_run_v2_service.gateway[0].template[0].containers[1].args[0], "REDIS_SSL_CA_CERTS") &&
|
||||
{ for e in google_cloud_run_v2_service.gateway[0].template[0].containers[1].env : e.name => e.value }["LITELLM_JOB_ROLE"] == "collector"
|
||||
)
|
||||
error_message = "The sidecar must run litellm.proxy.collector from the gateway image with the same Redis CA + DATABASE_URL bootstrap as the gateway."
|
||||
error_message = "The sidecar must run the collector component of the gateway's image with the same Redis CA + DATABASE_URL bootstrap as the gateway."
|
||||
}
|
||||
|
||||
assert {
|
||||
|
|
|
|||
|
|
@ -144,17 +144,15 @@ run "gateway_starts_through_the_pool_aware_launcher" {
|
|||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
strcontains(local.gateway_launch_cmd, "exec python -m gateway.launch --host 0.0.0.0 --port 4000 --workers 4"),
|
||||
strcontains(local.gateway_launch_cmd, "exec ddtrace-run python -m gateway.launch --host 0.0.0.0 --port 4000 --workers 4"),
|
||||
!strcontains(local.gateway_launch_cmd, "uvicorn gateway.main:app"),
|
||||
endswith(google_cloud_run_v2_service.gateway[0].template[0].containers[0].args[0], local.gateway_launch_cmd),
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[0].command == tolist(["sh", "-c"]),
|
||||
endswith(google_cloud_run_v2_service.gateway[0].template[0].containers[0].args[0], " && exec /app/docker-entrypoint.sh gateway --workers 4"),
|
||||
])
|
||||
error_message = "The gateway must start through gateway.launch (with and without ddtrace) so the pooler starts once before uvicorn forks the workers."
|
||||
error_message = "The gateway must start through the image's gateway component with the configured worker count so the pooler starts once before uvicorn forks the workers."
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = strcontains(local.backend_launch_cmd, "uvicorn backend.main:app")
|
||||
error_message = "The backend has no workers to share a pooler and keeps starting uvicorn directly."
|
||||
condition = endswith(google_cloud_run_v2_service.backend[0].template[0].containers[0].args[0], " && exec /app/docker-entrypoint.sh backend")
|
||||
error_message = "The backend has no workers to share a pooler and starts the plain backend component."
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -65,11 +65,11 @@ run "metrics_sidecar_enabled" {
|
|||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].image == local.gateway_image,
|
||||
join(" ", google_cloud_run_v2_service.gateway[0].template[0].containers[1].command) == "python -m litellm.proxy.prometheus_metrics_server",
|
||||
join(" ", google_cloud_run_v2_service.gateway[0].template[0].containers[1].args) == "--port 4001",
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].image == google_cloud_run_v2_service.gateway[0].template[0].containers[0].image,
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[1].command == null,
|
||||
join(" ", google_cloud_run_v2_service.gateway[0].template[0].containers[1].args) == "metrics --port 4001",
|
||||
])
|
||||
error_message = "The metrics sidecar must run the gateway image's prometheus_metrics_server on the configured port."
|
||||
error_message = "The metrics sidecar must run the gateway image's metrics component on the configured port."
|
||||
}
|
||||
|
||||
assert {
|
||||
|
|
|
|||
76
terraform/litellm/gcp/tests/unified_image.tftest.hcl
Normal file
76
terraform/litellm/gcp/tests/unified_image.tftest.hcl
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
# Plan-only coverage for the single-image contract: every Cloud Run service
|
||||
# and the migrations Job pull the same image and name their process through
|
||||
# the image entrypoint's component argument.
|
||||
|
||||
mock_provider "google" {
|
||||
mock_resource "google_redis_instance" {
|
||||
defaults = {
|
||||
host = "10.0.0.4"
|
||||
port = 6379
|
||||
server_ca_certs = [{
|
||||
cert = "-----BEGIN CERTIFICATE-----\nmock\n-----END CERTIFICATE-----"
|
||||
}]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mock_provider "google-beta" {}
|
||||
mock_provider "random" {}
|
||||
|
||||
variables {
|
||||
project_id = "test-project"
|
||||
tenant = "tenant"
|
||||
env = "test"
|
||||
allow_plaintext_lb = true
|
||||
image_registry = "us-central1-docker.pkg.dev/test-project/litellm/berriai"
|
||||
image_tag = "v9.9.9-stable"
|
||||
gateway_num_workers = 3
|
||||
}
|
||||
|
||||
run "registry_and_tag_compose_one_image_for_every_workload" {
|
||||
command = plan
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[0].image == "us-central1-docker.pkg.dev/test-project/litellm/berriai/litellm:v9.9.9-stable",
|
||||
google_cloud_run_v2_service.backend[0].template[0].containers[0].image == "us-central1-docker.pkg.dev/test-project/litellm/berriai/litellm:v9.9.9-stable",
|
||||
google_cloud_run_v2_service.ui[0].template[0].containers[0].image == "us-central1-docker.pkg.dev/test-project/litellm/berriai/litellm:v9.9.9-stable",
|
||||
google_cloud_run_v2_job.migrations[0].template[0].template[0].containers[0].image == "us-central1-docker.pkg.dev/test-project/litellm/berriai/litellm:v9.9.9-stable",
|
||||
terraform_data.migration[0].triggers_replace.job_image == "us-central1-docker.pkg.dev/test-project/litellm/berriai/litellm:v9.9.9-stable",
|
||||
])
|
||||
error_message = "All workloads must pull <image_registry>/litellm:<image_tag> and the migration re-run trigger must follow it."
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
google_cloud_run_v2_service.ui[0].template[0].containers[0].command == null,
|
||||
google_cloud_run_v2_service.ui[0].template[0].containers[0].args == tolist(["ui"]),
|
||||
google_cloud_run_v2_job.migrations[0].template[0].template[0].containers[0].command == null,
|
||||
google_cloud_run_v2_job.migrations[0].template[0].template[0].containers[0].args == tolist(["migrations"]),
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[0].command == tolist(["sh", "-c"]),
|
||||
endswith(google_cloud_run_v2_service.gateway[0].template[0].containers[0].args[0], " && exec /app/docker-entrypoint.sh gateway --workers 3"),
|
||||
google_cloud_run_v2_service.backend[0].template[0].containers[0].command == tolist(["sh", "-c"]),
|
||||
endswith(google_cloud_run_v2_service.backend[0].template[0].containers[0].args[0], " && exec /app/docker-entrypoint.sh backend"),
|
||||
])
|
||||
error_message = "UI and migrations must keep the image entrypoint with their component as args; gateway and backend must exec it with their component after the DATABASE_URL bootstrap."
|
||||
}
|
||||
}
|
||||
|
||||
run "image_overrides_the_composed_uri_everywhere" {
|
||||
command = plan
|
||||
|
||||
variables {
|
||||
image = "us-central1-docker.pkg.dev/test-project/mirror/litellm:custom"
|
||||
}
|
||||
|
||||
assert {
|
||||
condition = alltrue([
|
||||
google_cloud_run_v2_service.gateway[0].template[0].containers[0].image == var.image,
|
||||
google_cloud_run_v2_service.backend[0].template[0].containers[0].image == var.image,
|
||||
google_cloud_run_v2_service.ui[0].template[0].containers[0].image == var.image,
|
||||
google_cloud_run_v2_job.migrations[0].template[0].template[0].containers[0].image == var.image,
|
||||
terraform_data.migration[0].triggers_replace.job_image == var.image,
|
||||
])
|
||||
error_message = "A full image URI must win over image_registry + image_tag for every workload."
|
||||
}
|
||||
}
|
||||
|
|
@ -119,63 +119,43 @@ variable "vpc_connector_cidr" {
|
|||
default = "10.41.0.0/28"
|
||||
}
|
||||
|
||||
# ---------- Component images ----------
|
||||
# ---------- Image ----------
|
||||
#
|
||||
# Cloud Run only pulls from Artifact Registry, [region.]gcr.io, or
|
||||
# docker.io — it rejects arbitrary registries (notably ghcr.io) at apply
|
||||
# time. The four images live on GHCR upstream, so any real deploy must
|
||||
# either set `image_registry` to an Artifact Registry remote repository
|
||||
# pointed at ghcr.io (e.g. `us-central1-docker.pkg.dev/my-proj/litellm/berriai`)
|
||||
# or override the per-component `*_image` vars individually with full URIs.
|
||||
# docker.io and rejects arbitrary registries (notably ghcr.io) at apply
|
||||
# time. The image lives on GHCR upstream, so any real deploy must either
|
||||
# set `image_registry` to an Artifact Registry remote repository pointed at
|
||||
# ghcr.io (e.g. `us-central1-docker.pkg.dev/my-proj/litellm/berriai`) or
|
||||
# set `image` to the full URI of a copy you mirrored yourself.
|
||||
|
||||
variable "image_registry" {
|
||||
description = <<-EOT
|
||||
Registry path prefix used to compose the four LiteLLM image URIs as
|
||||
`<image_registry>/litellm-<component>:<image_tag>`. The default
|
||||
(`ghcr.io/berriai`) only works on registries Cloud Run accepts — for
|
||||
GHCR-backed deploys, create an Artifact Registry remote repository
|
||||
pointed at `https://ghcr.io` and set this to that repo's path
|
||||
Registry path prefix used to compose the LiteLLM image URI as
|
||||
`<image_registry>/litellm:<image_tag>`. The default (`ghcr.io/berriai`)
|
||||
only works on registries Cloud Run accepts: for GHCR-backed deploys,
|
||||
create an Artifact Registry remote repository pointed at
|
||||
`https://ghcr.io` and set this to that repo's path
|
||||
(e.g. `us-central1-docker.pkg.dev/<project>/<remote-repo>/berriai`).
|
||||
Per-component overrides (`gateway_image`, `backend_image`, `ui_image`,
|
||||
`migrations_image`) bypass this entirely when set.
|
||||
`image` bypasses this entirely when set.
|
||||
EOT
|
||||
type = string
|
||||
default = "ghcr.io/berriai"
|
||||
}
|
||||
|
||||
variable "image_tag" {
|
||||
description = "Tag applied to all four litellm-* images when composed from `image_registry`. Bump in lockstep when bumping LiteLLM. Must match a tag actually published to GHCR — the split images use the `v`-prefixed semver convention (e.g. `v1.86.0-dev`)."
|
||||
description = "Tag of the LiteLLM image when composed from `image_registry`. The component entrypoint ships from v1.104.0; older tags only run the monolithic proxy. Must match a tag actually published to GHCR (`v<semver>` for releases)."
|
||||
type = string
|
||||
default = "v1.86.0-dev"
|
||||
default = "v1.104.0"
|
||||
}
|
||||
|
||||
variable "gateway_image" {
|
||||
description = "Full image URI for the gateway. Empty (default) composes from `image_registry` + `image_tag`. Public images or Artifact Registry only — Cloud Run won't authenticate against arbitrary private registries."
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "backend_image" {
|
||||
description = "Full image URI for the backend. Empty (default) composes from `image_registry` + `image_tag`."
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "ui_image" {
|
||||
description = "Full image URI for the UI. Empty (default) composes from `image_registry` + `image_tag`."
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "migrations_image" {
|
||||
variable "image" {
|
||||
description = <<-EOT
|
||||
Full image URI for the one-off prisma migration Cloud Run Job. Empty
|
||||
(default) composes from `image_registry` + `image_tag` as
|
||||
`litellm-migrations`. Built from `migrations/Dockerfile` — slim image
|
||||
whose ENTRYPOINT runs `python3 /app/run.py` (assembles DATABASE_URL
|
||||
from DATABASE_* env vars via DatabaseURLSettings, then runs
|
||||
`prisma migrate deploy`). Should track the same release tag as
|
||||
gateway/backend/ui.
|
||||
Full URI of the one LiteLLM image every Cloud Run service and the
|
||||
migrations Job run. Its entrypoint picks the process from the first
|
||||
argument (`gateway`, `backend`, `ui`, `migrations`, `metrics`,
|
||||
`collector`). Empty (default) composes from `image_registry` +
|
||||
`image_tag`. Public images or Artifact Registry only: Cloud Run won't
|
||||
authenticate against arbitrary private registries.
|
||||
EOT
|
||||
type = string
|
||||
default = ""
|
||||
|
|
@ -563,8 +543,7 @@ variable "gateway_metrics_port" {
|
|||
Serve Prometheus /metrics from a `metrics` sidecar container in the
|
||||
gateway Cloud Run service on this port (a whole number 1-65535, not 4000
|
||||
or 13133), so the collector's scrape never runs on an inference worker.
|
||||
The sidecar runs the gateway image with
|
||||
`python -m litellm.proxy.prometheus_metrics_server` and aggregates the
|
||||
The sidecar runs the `metrics` component of the image and aggregates the
|
||||
workers' PROMETHEUS_MULTIPROC_DIR samples over an in-memory volume shared
|
||||
with the gateway container. Cloud Run only routes ingress to the gateway
|
||||
container, so the sidecar port is reachable on localhost inside the
|
||||
|
|
@ -572,7 +551,7 @@ variable "gateway_metrics_port" {
|
|||
(gateway_metrics_collector_image) scrapes it and writes the series to
|
||||
Cloud Monitoring. The load balancer keeps serving the authenticated
|
||||
/metrics on the gateway port as before. Null (the default) leaves /metrics
|
||||
on the gateway port only. Needs gateway_image v1.101.0 or newer.
|
||||
on the gateway port only.
|
||||
EOT
|
||||
type = number
|
||||
default = null
|
||||
|
|
@ -660,7 +639,7 @@ variable "billing_metrics_ca_cert_pem" {
|
|||
# ---------- Collector sidecar ----------
|
||||
#
|
||||
# Opt-in offload of spend tracking from the gateway's uvicorn workers to a
|
||||
# `python -m litellm.proxy.collector` sidecar container in the same Cloud Run
|
||||
# `collector` component sidecar container in the same Cloud Run
|
||||
# instance (helm's `gateway.collector`, mirrors the AWS stack). Containers
|
||||
# in one instance share localhost, so the sidecar listens on loopback TCP.
|
||||
# Disabled (the default) adds nothing to the service.
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
"""Unit tests for `docker-entrypoint.sh`, the component dispatcher every shipped
|
||||
container runs, and for the Terraform launch commands that must agree with it."""
|
||||
container runs, and for the component arguments the Terraform modules hand it."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
|
@ -13,8 +13,10 @@ import pytest
|
|||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
DOCKER_ENTRYPOINT = REPO_ROOT / "docker-entrypoint.sh"
|
||||
DOCKERFILE = REPO_ROOT / "Dockerfile"
|
||||
TERRAFORM_ECS = REPO_ROOT / "terraform" / "litellm" / "aws" / "ecs.tf"
|
||||
TERRAFORM_CLOUDRUN = REPO_ROOT / "terraform" / "litellm" / "gcp" / "cloudrun.tf"
|
||||
TERRAFORM_MODULES = {
|
||||
"aws": REPO_ROOT / "terraform" / "litellm" / "aws",
|
||||
"gcp": REPO_ROOT / "terraform" / "litellm" / "gcp",
|
||||
}
|
||||
|
||||
IMAGE_ENTRYPOINT_PATH = "/app/docker-entrypoint.sh"
|
||||
STUBBED_EXECUTABLES = ("ddtrace-run", "uvicorn", "python", "litellm", "nginx")
|
||||
|
|
@ -35,17 +37,25 @@ _STUB_TEMPLATE = """#!/bin/sh
|
|||
|
||||
_ENTRYPOINT_RE = re.compile(r"^ENTRYPOINT\s+(\[.*\])\s*$", re.MULTILINE)
|
||||
_CMD_RE = re.compile(r"^CMD\s+(\[.*\])\s*$", re.MULTILINE)
|
||||
_APP_TARGET_RE = re.compile(r"(?:gateway|backend)\.main:app|gateway\.launch")
|
||||
_TF_STRING_LOCAL_RE = re.compile(r'^\s*(\w+)\s*=\s*"((?:[^"\\]|\\.)*)"\s*$', re.MULTILINE)
|
||||
_TF_INTERPOLATION_RE = re.compile(r"\$\{(local|var)\.(\w+)\}")
|
||||
_TF_ENTRYPOINT_LOCAL_RE = re.compile(r'^\s*image_entrypoint\s*=\s*"([^"]*)"\s*$', re.MULTILINE)
|
||||
_TF_WORD = r'(?:"[^"\s]*"|tostring\(var\.\w+\)|\$\{var\.\w+\}|[\w.-]+)'
|
||||
_TF_ENTRYPOINT_CALL_RE = re.compile(rf"exec \$\{{local\.image_entrypoint\}}((?: {_TF_WORD})+)")
|
||||
_TF_COMPONENT_LIST_RE = re.compile(
|
||||
rf"(?:command|args|gateway|backend|ui|migrations|metrics|collector)\s*=\s*"
|
||||
rf'\[\s*("(?:proxy|gateway|backend|ui|migrations|metrics|collector)"(?:\s*,\s*{_TF_WORD})*)\s*\]'
|
||||
)
|
||||
_TF_VAR_RE = re.compile(r"tostring\(var\.(\w+)\)|\$\{var\.(\w+)\}")
|
||||
|
||||
TERRAFORM_LAUNCH_SITES = {TERRAFORM_ECS: 2, TERRAFORM_CLOUDRUN: 2}
|
||||
TERRAFORM_VAR_STUBS = {"gateway_num_workers": "2"}
|
||||
COMPONENT_LAUNCHERS = {
|
||||
"gateway": ("python", "-m", "gateway.launch"),
|
||||
"backend": ("uvicorn", "backend.main:app"),
|
||||
TERRAFORM_VAR_STUBS = {"gateway_num_workers": "2", "gateway_metrics_port": "9464"}
|
||||
TERRAFORM_WORKLOADS = frozenset({"gateway", "backend", "ui", "migrations", "metrics", "collector"})
|
||||
COMPONENT_BINARIES = {
|
||||
"gateway": "python",
|
||||
"backend": "uvicorn",
|
||||
"ui": "nginx",
|
||||
"migrations": "python",
|
||||
"metrics": "python",
|
||||
"collector": "python",
|
||||
}
|
||||
_MAX_INTERPOLATION_PASSES = 5
|
||||
|
||||
|
||||
def _write_stubs(bin_dir: Path, names: tuple[str, ...]) -> None:
|
||||
|
|
@ -85,30 +95,24 @@ def _run_entrypoint(
|
|||
return tuple(record.read_text().splitlines()) if record.exists() else ()
|
||||
|
||||
|
||||
def _run_shell_command(command: str, bin_dir: Path, record: Path, use_ddtrace: str | None) -> tuple[str, ...]:
|
||||
"""Run a resolved Terraform launch command through `sh -c` and return the recorded lines."""
|
||||
env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace})
|
||||
result = subprocess.run(["sh", "-c", command], env=env, capture_output=True, text=True, check=False)
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
return tuple(record.read_text().splitlines()) if record.exists() else ()
|
||||
def _tf_word(word: str) -> str:
|
||||
"""Turn one HCL argument token into the string the container receives, stubbing `var.` references."""
|
||||
var = _TF_VAR_RE.fullmatch(word)
|
||||
if var is not None:
|
||||
return TERRAFORM_VAR_STUBS[var.group(1) or var.group(2)]
|
||||
return word.strip('"')
|
||||
|
||||
|
||||
def _resolve_tf_local(terraform_file: Path, name: str) -> str:
|
||||
"""Read a string local out of a .tf file and expand its `local.` / `var.` interpolations."""
|
||||
values = {
|
||||
m.group(1): m.group(2).replace('\\"', '"') for m in _TF_STRING_LOCAL_RE.finditer(terraform_file.read_text())
|
||||
}
|
||||
|
||||
assert name in values, f"{terraform_file} defines no {name} local"
|
||||
resolved = values[name]
|
||||
for _ in range(_MAX_INTERPOLATION_PASSES):
|
||||
if "${" not in resolved:
|
||||
return resolved
|
||||
resolved = _TF_INTERPOLATION_RE.sub(
|
||||
lambda m: values[m.group(2)] if m.group(1) == "local" else TERRAFORM_VAR_STUBS[m.group(2)],
|
||||
resolved,
|
||||
def _terraform_component_argvs(module_dir: Path) -> tuple[tuple[str, tuple[str, ...]], ...]:
|
||||
"""Every (file, argv) a module hands the image entrypoint, whether as a `command`/`args` list or an `exec` string."""
|
||||
return tuple(
|
||||
(tf.name, tuple(_tf_word(word) for word in words))
|
||||
for tf in sorted(module_dir.glob("*.tf"))
|
||||
for words in (
|
||||
*(m.group(1).split() for m in _TF_ENTRYPOINT_CALL_RE.finditer(tf.read_text())),
|
||||
*(re.split(r"\s*,\s*", m.group(1)) for m in _TF_COMPONENT_LIST_RE.finditer(tf.read_text())),
|
||||
)
|
||||
raise AssertionError(f"{terraform_file}:{name} still has unresolved interpolations: {resolved}")
|
||||
) # comprehension-ok: the two regex scans over one file read are clearer inline than as a generator helper
|
||||
|
||||
|
||||
def _entrypoint_argv(dockerfile: Path) -> tuple[str, ...]:
|
||||
|
|
@ -382,81 +386,40 @@ def test_creates_a_missing_prometheus_multiproc_dir(tmp_path: Path) -> None:
|
|||
assert missing.is_dir()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name)
|
||||
@pytest.mark.parametrize("component", ["gateway", "backend"])
|
||||
@pytest.mark.parametrize("use_ddtrace", [*TRUTHY_USE_DDTRACE, *FALSY_USE_DDTRACE])
|
||||
def test_terraform_launch_command_matches_the_script_contract(
|
||||
terraform_file: Path, component: str, use_ddtrace: str | None, tmp_path: Path
|
||||
) -> None:
|
||||
"""The Terraform command and `docker-entrypoint.sh <component>` must decide identically.
|
||||
@pytest.mark.parametrize("module", TERRAFORM_MODULES, ids=str)
|
||||
def test_terraform_hands_every_workload_to_the_image_entrypoint(module: str) -> None:
|
||||
"""Both modules run each workload as a component of the one image, and nothing else.
|
||||
|
||||
The decision deliberately lives in two places. The script is what the image ENTRYPOINT runs;
|
||||
the Terraform strings are what runs when a deployment overrides that ENTRYPOINT, and they
|
||||
cannot call the script because the caller supplies the image tag and it may predate the file.
|
||||
So instead of asserting a shared path, this runs both implementations under the same
|
||||
environment and asserts they agree on which binary is exec'd and on whether the openai
|
||||
integration is disabled.
|
||||
The set is pinned so a workload that stops naming its component (falling back to the proxy)
|
||||
or one that execs an entrypoint path the image does not ship fails here instead of at container start.
|
||||
"""
|
||||
launcher = COMPONENT_LAUNCHERS[component]
|
||||
app_target = " ".join(launcher[1:])
|
||||
command = _resolve_tf_local(terraform_file, f"{component}_launch_cmd")
|
||||
argvs = _terraform_component_argvs(TERRAFORM_MODULES[module])
|
||||
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir(parents=True)
|
||||
_write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python"))
|
||||
from_terraform = _run_shell_command(command, bin_dir, tmp_path / "terraform.txt", use_ddtrace)
|
||||
|
||||
from_script = _run_entrypoint((component,), tmp_path / "script", use_ddtrace=use_ddtrace)
|
||||
|
||||
assert from_terraform[0] == from_script[0], (
|
||||
f"{terraform_file} disagrees with the script on USE_DDTRACE={use_ddtrace}"
|
||||
)
|
||||
assert from_terraform[2] == from_script[2], f"{terraform_file} disagrees with the script on the openai integration"
|
||||
assert app_target in from_terraform[1]
|
||||
assert app_target in from_script[1]
|
||||
assert "gateway.main:app" not in from_terraform[1], f"{terraform_file} bypasses the gateway.launch supervisor"
|
||||
|
||||
if use_ddtrace in TRUTHY_USE_DDTRACE:
|
||||
assert from_terraform[0] == "exec=ddtrace-run"
|
||||
assert from_terraform[1].startswith(f"args={launcher[0]} ")
|
||||
assert from_terraform[2] == "DD_TRACE_OPENAI_ENABLED=False"
|
||||
else:
|
||||
assert from_terraform[0] == f"exec={launcher[0]}"
|
||||
assert from_terraform[2] == "DD_TRACE_OPENAI_ENABLED=<unset>"
|
||||
assert {argv[0] for _, argv in argvs} == TERRAFORM_WORKLOADS, f"{module} workloads: {argvs}"
|
||||
for tf in TERRAFORM_MODULES[module].glob("*.tf"):
|
||||
for entrypoint_path in _TF_ENTRYPOINT_LOCAL_RE.findall(tf.read_text()):
|
||||
assert entrypoint_path == _entrypoint_argv(DOCKERFILE)[0], f"{tf} execs a path the image does not ship"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name)
|
||||
def test_terraform_routes_every_launch_site_through_a_traced_command(terraform_file: Path) -> None:
|
||||
"""Every place Terraform names a component ASGI target has to honor `USE_DDTRACE`.
|
||||
@pytest.mark.parametrize("module", TERRAFORM_MODULES, ids=str)
|
||||
@pytest.mark.parametrize("use_ddtrace", ["true", None])
|
||||
def test_the_entrypoint_accepts_every_terraform_argv(module: str, use_ddtrace: str | None, tmp_path: Path) -> None:
|
||||
"""Each argv Terraform passes must select the intended process and forward its extra flags.
|
||||
|
||||
The count is pinned as well: a launch site that is deleted or renamed would otherwise drop
|
||||
out of the scan and let this pass while covering less than it claims.
|
||||
Runs the real script under every module's arguments, with `var.` references stubbed, so a
|
||||
renamed component or a flag the entrypoint swallows shows up as the wrong binary or lost args.
|
||||
"""
|
||||
launch_sites = tuple(line for line in terraform_file.read_text().splitlines() if _APP_TARGET_RE.search(line))
|
||||
for index, (tf, argv) in enumerate(_terraform_component_argvs(TERRAFORM_MODULES[module])):
|
||||
component, *extra = argv
|
||||
recorded = _run_entrypoint(argv, tmp_path / str(index), use_ddtrace=use_ddtrace)
|
||||
|
||||
assert len(launch_sites) == TERRAFORM_LAUNCH_SITES[terraform_file], (
|
||||
f"{terraform_file} launch-site count changed; re-check each one honors USE_DDTRACE"
|
||||
)
|
||||
for line in launch_sites:
|
||||
assert "ddtrace-run" in line, f"{terraform_file} launches uvicorn without honoring USE_DDTRACE: {line.strip()}"
|
||||
|
||||
body = terraform_file.read_text()
|
||||
for component in ("gateway", "backend"):
|
||||
assert f"local.{component}_launch_cmd" in body, (
|
||||
f"{terraform_file} defines a {component} launch command but never uses it"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name)
|
||||
def test_terraform_does_not_depend_on_the_entrypoint_script(terraform_file: Path) -> None:
|
||||
"""Terraform must not reference a file the caller-supplied image may not contain.
|
||||
|
||||
Both modules default to an image tag published before the script existed, so exec'ing that
|
||||
path would fail at container start rather than degrade to an untraced process.
|
||||
"""
|
||||
assert IMAGE_ENTRYPOINT_PATH not in terraform_file.read_text(), (
|
||||
f"{terraform_file} depends on a script that images predating it do not ship"
|
||||
)
|
||||
expected_exec = "ddtrace-run" if use_ddtrace and component != "ui" else COMPONENT_BINARIES[component]
|
||||
assert recorded[0] == f"exec={expected_exec}", f"{module}/{tf} {argv}: {recorded}"
|
||||
assert recorded[1].endswith(" ".join(extra)), f"{module}/{tf} {argv} drops its flags: {recorded[1]}"
|
||||
if component == "gateway":
|
||||
assert extra == ["--workers", TERRAFORM_VAR_STUBS["gateway_num_workers"]], f"{module}/{tf}: {argv}"
|
||||
if component == "metrics":
|
||||
assert extra == ["--port", TERRAFORM_VAR_STUBS["gateway_metrics_port"]], f"{module}/{tf}: {argv}"
|
||||
|
||||
|
||||
def test_entrypoint_script_is_executable() -> None:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue