This commit is contained in:
devin-ai-integration[bot] 2026-09-30 22:18:00 +00:00 • committed by GitHub
commit 8b3782e7d7
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
43 changed files with 738 additions and 1765 deletions

View file

@ -532,11 +532,10 @@ jobs:
key: v1-uv-cache-{{ checksum "uv.lock" }}
- save_cargo_target
- run:
name: Run prisma ./docker/entrypoint.sh
name: Run prisma migrations
command: |
set +e
chmod +x docker/entrypoint.sh
./docker/entrypoint.sh
uv run --no-sync python litellm/proxy/prisma_migration.py
set -e
# Run pytest and generate JUnit XML report
- run:
@ -606,11 +605,10 @@ jobs:
- ~/.cache/uv
key: v1-uv-cache-{{ checksum "uv.lock" }}
- run:
name: Run prisma ./docker/entrypoint.sh
name: Run prisma migrations
command: |
set +e
chmod +x docker/entrypoint.sh
./docker/entrypoint.sh
uv run --no-sync python litellm/proxy/prisma_migration.py
set -e
# Run pytest and generate JUnit XML report
- run:
@ -681,11 +679,10 @@ jobs:
- ~/.cache/uv
key: v1-uv-cache-{{ checksum "uv.lock" }}
- run:
name: Run prisma ./docker/entrypoint.sh
name: Run prisma migrations
command: |
set +e
chmod +x docker/entrypoint.sh
./docker/entrypoint.sh
uv run --no-sync python litellm/proxy/prisma_migration.py
set -e
# Run pytest and generate JUnit XML report
@ -2422,16 +2419,18 @@ jobs:
path: test-results
proxy_build_from_pip_tests:
# Change from docker to machine executor
# Validates the published PyPI artifact, not the checked-out source tree: the
# proxy is pip-installed from PyPI into its own venv and booted on the runner
machine:
image: ubuntu-2204:2024.04.1
resource_class: large
working_directory: ~/project
environment:
LITELLM_PIP_VERSION: "1.83.0"
steps:
- checkout
- skip_if_unrelated_changes
- setup_google_dns
# Remove Docker CLI installation since it's already available in machine executor
- install_uv
- install_rust
- run:
@ -2439,47 +2438,53 @@ jobs:
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- run:
name: Build Docker image
name: Install the published litellm proxy from PyPI into its own venv
command: |
docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip .
uv venv --python 3.13 /tmp/litellm-pip
uv pip install --python /tmp/litellm-pip/bin/python \
"litellm[proxy,proxy-runtime]==${LITELLM_PIP_VERSION}" \
"google-cloud-aiplatform==1.133.0" \
"google-genai==1.37.0" \
"anthropic[vertex]==0.84.0" \
"grpcio==1.78.0" \
"prometheus-client==0.20.0" \
"langfuse==2.59.7" \
"opentelemetry-api==1.28.0" \
"opentelemetry-sdk==1.28.0" \
"opentelemetry-exporter-otlp==1.28.0" \
"ddtrace==4.11.0" \
"sentry-sdk==2.21.0" \
"mangum==0.17.0" \
"azure-ai-contentsafety==1.0.0" \
"azure-storage-file-datalake==12.20.0" \
"pypdf==6.7.5" \
"llm-sandbox==0.3.31" \
"detect-secrets==1.5.0" \
"prisma==0.11.0" \
"openai==2.24.0"
/tmp/litellm-pip/bin/python -c "import litellm, sys; print('litellm', litellm.__version__ if hasattr(litellm, '__version__') else '', sys.version)"
- start_postgres
- start_fake_openai_endpoint
- run:
name: Run Docker container
name: Run the published proxy
# intentionally give bad redis credentials here
# the OTEL test - should get this as a trace
command: |
docker run -d \
-p 4000:4000 \
-e LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true \
-e DATABASE_URL=postgresql://postgres:postgres@host.docker.internal:5432/circle_test \
-e REDIS_HOST=$REDIS_HOST \
-e REDIS_PASSWORD=$REDIS_PASSWORD \
-e REDIS_PORT=$REDIS_PORT \
-e LITELLM_MASTER_KEY="sk-1234" \
-e OPENAI_API_KEY=$OPENAI_API_KEY \
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
-e LITELLM_LICENSE=$LITELLM_LICENSE \
-e OTEL_EXPORTER="in_memory" \
-e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \
-e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
-e AWS_REGION_NAME=$AWS_REGION_NAME \
-e COHERE_API_KEY=$COHERE_API_KEY \
-e USE_DDTRACE=True \
-e DD_API_KEY=$DD_API_KEY \
-e DD_SITE=$DD_SITE \
-e GCS_FLUSH_INTERVAL="1" \
-e LITELLM_LOG=ERROR \
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \
my-app:latest \
--config /app/config.yaml \
--port 4000
- run:
name: Start outputting logs
command: docker logs -f my-app
background: true
command: |
export LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true
export DATABASE_URL=postgresql://postgres:postgres@localhost:5432/circle_test
export LITELLM_MASTER_KEY="sk-1234"
export LITELLM_LICENSE="bad-license"
export FAKE_OPENAI_API_BASE=http://localhost:8190
export OTEL_EXPORTER="in_memory"
export USE_DDTRACE=True
export DD_TRACE_OPENAI_ENABLED="False"
export GCS_FLUSH_INTERVAL="1"
export LITELLM_LOG=ERROR
cd /tmp/litellm-pip
exec /tmp/litellm-pip/bin/ddtrace-run /tmp/litellm-pip/bin/litellm \
--config ~/project/tests/basic_proxy_startup_tests/build_from_pip_config.yaml \
--port 4000 2>&1 | tee /tmp/litellm-pip.log
- wait_for_service:
url: http://localhost:4000
timeout: "300"
@ -2495,14 +2500,15 @@ jobs:
--junitxml=test-results/junit-2.xml \
--durations=5"
no_output_timeout: 15m
# Clean up first container
- store_test_results:
path: test-results
- run:
name: Stop and remove first container
name: Proxy log
command: cat /tmp/litellm-pip.log || true
when: always
- run:
name: Stop postgres
command: |
docker stop my-app || true
docker rm my-app || true
docker stop postgres-db || true
docker rm postgres-db || true
when: always
@ -3023,7 +3029,7 @@ jobs:
docker build \
--label org.opencontainers.image.revision="$(git rev-parse HEAD)" \
-t litellm-docker-database:ci \
-f docker/Dockerfile.database .
-f Dockerfile .
fi
python3 .circleci/scripts/run_migration_tests.py record-image

View file

@ -7,7 +7,6 @@ tests
.devcontainer
*.tgz
log.txt
docker/Dockerfile.*
# Claude Flow generated files (must be excluded from Docker build)
.claude/

View file

@ -102,15 +102,6 @@ test_paths:
- tests/integration/test_oci_proxy_integration.py
dockerfiles:
- reason: >-
The dashboard container is a static Next.js export served by nginx, and the dashboard build
and lint workflows already exercise that output, so building the image adds no signal about it
paths:
- ui/Dockerfile
- reason: >-
An example image under cookbook/ that is documentation rather than a shipped artifact
paths:
- cookbook/litellm-ollama-docker-image/Dockerfile
- reason: >-
The Rust gateway image compiles the whole workspace in release mode, which is too slow for
a per-pull-request job while the gateway binary is still being assembled; the Rust lint,

View file

@ -26,17 +26,24 @@ jobs:
with:
persist-credentials: false
- name: Build the Render cron image
run: docker build -f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix:${{ github.sha }} tests/e2e
- name: Resolve and install the Claude Code CLI as the cron user
- name: Install the pinned uv the cron image ships
env:
UV_VERSION: 0.10.9
UV_SHA256: 20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
run: |
docker run --rm compat-matrix:${{ github.sha }} bash -c '
set -euo pipefail
whoami
gh --version
uv --version
version="$(uv run --no-project --python 3.12 python /opt/litellm/tests/e2e/claude_code/pr_gate_version_resolver.py)"
/opt/litellm/tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" /tmp/claude-cli
/tmp/claude-cli/claude --version
'
set -euo pipefail
curl -fsSLo "$RUNNER_TEMP/uv.tar.gz" "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz"
echo "${UV_SHA256} $RUNNER_TEMP/uv.tar.gz" | sha256sum -c -
mkdir -p "$RUNNER_TEMP/bin"
tar -xzf "$RUNNER_TEMP/uv.tar.gz" -C "$RUNNER_TEMP/bin" --strip-components=1 uv-x86_64-unknown-linux-gnu/uv
echo "$RUNNER_TEMP/bin" >> "$GITHUB_PATH"
- name: Resolve and install the Claude Code CLI as an unprivileged user
run: |
set -euo pipefail
whoami
gh --version
uv --version
version="$(uv run --no-project --python 3.12 python tests/e2e/claude_code/pr_gate_version_resolver.py)"
tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" "$RUNNER_TEMP/claude-cli"
"$RUNNER_TEMP/claude-cli/claude" --version

View file

@ -8,21 +8,17 @@ on:
- "litellm_**"
paths:
- Dockerfile
- docker/Dockerfile.non_root
- migrations/Dockerfile
- .dockerignore
- docker-entrypoint.sh
- migrations/run.py
- gateway/Dockerfile
- gateway/main.py
- backend/Dockerfile
- gateway/launch.py
- backend/main.py
- docker/component_entrypoint.sh
- docker/entrypoint.sh
- litellm/proxy/prisma_migration.py
- litellm-proxy-extras/**
- tests/proxy_migration_tests/**
- uv.lock
- ui/litellm-dashboard/package-lock.json
- ui/Dockerfile
- ui/nginx.conf
- .github/workflows/image-scan.yml
- .grype.yaml
@ -36,9 +32,12 @@ concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
env:
IMAGE: litellm-image-scan:${{ github.sha }}
jobs:
image-scan:
name: image-scan
build:
name: build
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
@ -51,7 +50,92 @@ jobs:
with:
persist-credentials: false
# One image ships; every component below is a mode of it. Built once
# here and handed to the verify matrix so a component check never runs
# against a different build than the scan.
- name: Build the image
run: docker build -t "$IMAGE" .
- name: Save the image for the verify matrix
run: docker save "$IMAGE" | zstd -T0 -3 -o "$RUNNER_TEMP/litellm-image.tar.zst"
- uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1
with:
name: litellm-image
path: ${{ runner.temp }}/litellm-image.tar.zst
retention-days: 1
compression-level: 0
verify:
name: ${{ matrix.check }}
needs: build
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
contents: read
strategy:
fail-fast: false
matrix:
check: [scan, migrations, gateway, backend, ui]
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: litellm-image
path: ${{ runner.temp }}
- name: Load the image
run: zstd -d --stdout "$RUNNER_TEMP/litellm-image.tar.zst" | docker load
- name: Set up Python
if: matrix.check != 'scan'
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Install pytest
if: matrix.check != 'scan'
run: python -m pip install "pytest==9.0.3"
# The prisma bake must migrate a fresh DB with no egress as an arbitrary
# non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyFilesystem).
# `docker run` as the default uid with network hides a broken bake because
# the migration entrypoint exits 0 even when it applied nothing; asserting
# the schema was created is what catches it.
- name: Verify offline migration as a non-root uid
if: matrix.check == 'migrations'
env:
LITELLM_IMAGE: ${{ env.IMAGE }}
run: |
python -m pytest -v \
tests/proxy_migration_tests/test_offline_image_migration.py \
tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py
- name: Verify the gateway component serves offline as a non-root uid
if: matrix.check == 'gateway'
env:
LITELLM_IMAGE: ${{ env.IMAGE }}
LITELLM_COMPONENT: gateway
run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py
- name: Verify the backend component serves offline as a non-root uid
if: matrix.check == 'backend'
env:
LITELLM_IMAGE: ${{ env.IMAGE }}
LITELLM_COMPONENT: backend
run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py
- name: Verify the ui component serves as an arbitrary uid with a read-only root fs
if: matrix.check == 'ui'
env:
LITELLM_IMAGE: ${{ env.IMAGE }}
run: python -m pytest -v tests/proxy_migration_tests/test_ui_image_serves_offline.py
- name: Download Grype v0.114.0
if: matrix.check == 'scan'
run: |
curl -fsSL --retry 3 -o "$RUNNER_TEMP/grype.tar.gz" \
https://github.com/anchore/grype/releases/download/v0.114.0/grype_0.114.0_linux_amd64.tar.gz
@ -59,29 +143,6 @@ jobs:
tar xzf "$RUNNER_TEMP/grype.tar.gz" -C "$RUNNER_TEMP" grype
chmod +x "$RUNNER_TEMP/grype"
# Dockerfile.non_root is the rootless variant we ship. The other
# Dockerfiles share the same wolfi base and apk set, so OS-layer coverage
# is the same; matrix-scan if those variants ever diverge.
- name: Build runtime image
run: docker build -f docker/Dockerfile.non_root -t litellm-image-scan:${{ github.sha }} .
# The prisma bake must migrate a fresh DB with no egress as an arbitrary
# non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyRootFilesystem).
# `docker run` as the default uid with network hides a broken bake because
# the migration entrypoint exits 0 even when it applied nothing; asserting
# the schema was created is what catches it.
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify offline migration as a non-root uid
env:
LITELLM_IMAGE: litellm-image-scan:${{ github.sha }}
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
# Scans the whole shipped artifact: OS/apk plus every language package
# baked into the image, including ones no lockfile declares (e.g. prisma's
# vendored node engine) that osv-scan cannot see. osv-scan stays the fast
@ -89,160 +150,12 @@ jobs:
# free OSS, run as a pinned, checksum-verified binary; no GitHub Action
# dependency and no vendor SaaS callout.
- name: Scan image for fixable HIGH/CRITICAL CVEs
if: matrix.check == 'scan'
env:
GRYPE_MATCH_PYTHON_USING_CPES: "true"
run: |
"$RUNNER_TEMP/grype" litellm-image-scan:${{ github.sha }} \
"$RUNNER_TEMP/grype" "$IMAGE" \
--config .grype.yaml \
--only-fixed \
--fail-on high \
--output table
runtime-image:
name: runtime-image
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 30
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build runtime image
run: docker build -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify offline migration as a non-root uid
env:
LITELLM_IMAGE: litellm-runtime-scan:${{ github.sha }}
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
migrations-image:
name: migrations-image
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 30
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build migrations image
run: docker build -f migrations/Dockerfile -t litellm-migrations-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify offline migration as a non-root uid
env:
LITELLM_IMAGE: litellm-migrations-scan:${{ github.sha }}
LITELLM_MIGRATION_INTERPRETER: python3
LITELLM_MIGRATION_SCRIPT: /app/run.py
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py -v
gateway-image:
name: gateway-image
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 30
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build gateway image
run: docker build -f gateway/Dockerfile -t litellm-gateway-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify the gateway serves offline as a non-root uid
env:
LITELLM_IMAGE: litellm-gateway-scan:${{ github.sha }}
LITELLM_COMPONENT_PORT: "4000"
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
ui-image:
name: ui-image
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 30
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build UI image
run: docker build -f ui/Dockerfile -t litellm-ui-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify the UI serves offline as an arbitrary uid with a read-only root fs
env:
LITELLM_IMAGE: litellm-ui-scan:${{ github.sha }}
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_ui_image_serves_offline.py -v
backend-image:
name: backend-image
runs-on: ubuntu-latest
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 30
permissions:
contents: read
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build backend image
run: docker build -f backend/Dockerfile -t litellm-backend-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Verify the backend serves offline as a non-root uid
env:
LITELLM_IMAGE: litellm-backend-scan:${{ github.sha }}
LITELLM_COMPONENT_PORT: "4001"
run: |
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py -v

View file

@ -33,8 +33,8 @@ jobs:
# Built through the image stage rather than the checkout, because the
# stage copies ui/litellm-dashboard/ alone: an import reaching above the
# dashboard root resolves in a checkout and fails in every image we ship.
# Dockerfile, docker/Dockerfile.non_root and ui/Dockerfile share this
# stage verbatim, so building one covers all three.
# The shipped image serves this stage's output in every component, so
# building the stage here covers what ships.
- name: Build the dashboard as the shipped images build it
if: steps.changes.outputs.decision != 'skip'
run: docker build --target ui-builder -f Dockerfile .

View file

@ -265,8 +265,8 @@ uv run litellm --config your_config.yaml
If you want to build the Docker image yourself:
```bash
# Build using the non-root Dockerfile
docker build -f docker/Dockerfile.non_root -t litellm_dev .
# Build the image (runs as a non-root user by default)
docker build -t litellm_dev .
# Generate a master key. Requests send it as the bearer token
export LITELLM_MASTER_KEY="sk-$(openssl rand -hex 32)"

View file

@ -111,8 +111,12 @@ RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
RUN sed -i 's/\r$//' docker-entrypoint.sh && chmod +x docker-entrypoint.sh
RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
cp -r litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \
cp litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \
touch /var/lib/litellm/ui/.litellm_ui_ready
# Runtime stage
FROM $LITELLM_RUNTIME_IMAGE AS runtime
@ -125,23 +129,40 @@ USER root
# https://github.com/BerriAI/litellm/issues/33518
RUN echo "https://packages.wolfi.dev/os" >> /etc/apk/repositories
# node (without npm) is required by the prisma CLI at runtime
RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent
# node (without npm) is required by the prisma CLI at runtime; nginx serves the
# static admin UI in the `ui` component; libevent is pgbouncer's runtime.
RUN for i in 1 2 3; do \
apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libatomic libevent nginx && break; \
[ "$i" = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
WORKDIR /app
# Runtime writes land under /app/.cache, /var/lib/litellm and /tmp (group-0
# writable, so an arbitrary uid and a read-only root fs both work). Prisma CLI
# and engines are baked under /opt/prisma so `prisma migrate deploy` needs no
# npm and no network (#33650, #24554).
ENV PATH="/app/.venv/bin:${PATH}" \
PYTHONPATH="/app" \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1 \
HOME=/app \
LITELLM_NON_ROOT=true \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
PRISMA_HIDE_UPDATE_MESSAGE=1 \
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
PRISMA_OFFLINE_MODE=true
# Copy only what runtime needs. The application is installed inside the venv;
# the rest of the builder's /app is source and build metadata that must not
# ship (manifest-scanning tools attribute everything in it to this image).
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
COPY --from=builder /app/.venv /app/.venv
COPY --from=builder /app/docker /app/docker
COPY --from=builder /app/docker-entrypoint.sh /app/docker-entrypoint.sh
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
# enterprise/ is imported by source path at runtime (proxy_cli puts the
@ -149,21 +170,34 @@ COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/pr
# enterprise.enterprise_hooks from it)
COPY --from=builder /app/enterprise /app/enterprise
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every
# runtime uid can read and that no cache volume mount shadows. The paths are
# pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and recorded into the
# generated client at build time, so `prisma migrate deploy` on a fresh
# database needs no npm and no network access (#33650, #24554).
COPY --from=builder /app/gateway /app/gateway
COPY --from=builder /app/backend /app/backend
COPY --from=builder /app/migrations /app/migrations
COPY --from=builder /opt/prisma /opt/prisma
COPY --from=builder /var/lib/litellm /var/lib/litellm
COPY ui/nginx.conf /etc/nginx/nginx.conf
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
find /app/.venv -type d -path "*/tornado/test" -delete && \
RUN find /app/.venv -depth -type d -path "*/tornado/test" -exec rm -rf {} + && \
mkdir -p /app/.cache && \
chown -R 65532:0 /app /var/lib/litellm && \
chmod -R g=u,g+w /app/.cache /var/lib/litellm && \
PRISMA_PATH="$(python -c 'import os, prisma; print(os.path.dirname(prisma.__file__))')" && \
PROXY_EXTRAS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))')" && \
chmod -R g=u,g+w "$PRISMA_PATH" "$PROXY_EXTRAS_PATH" && \
chmod -R a+rX /opt/prisma && \
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \
python -c "from litellm.rust_bridge.loader import native_bridge_available; assert native_bridge_available()" && \
python -c "import gateway.launch, backend.main"
EXPOSE 4000/tcp
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The
# numeric form is what the kubelet's runAsNonRoot admission check can verify.
USER 65532:65532
ENTRYPOINT ["docker/prod_entrypoint.sh"]
CMD ["--port", "4000"]
RUN nginx -t
EXPOSE 4000/tcp 4001/tcp 3000/tcp
ENTRYPOINT ["/app/docker-entrypoint.sh"]

View file

@ -1,106 +0,0 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin
# ---------- Builder ----------
FROM $LITELLM_BUILD_IMAGE AS builder
WORKDIR /app
USER root
COPY --from=uvbin /uv /uvx /usr/local/bin/
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
RUN for i in 1 2 3; do \
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start.
# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a
# BuildKit cache mount (different filesystem).
# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of
# silently pulling a managed interpreter.
# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't
# silently re-enable nodeenv's Node download.
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
UV_COMPILE_BYTECODE=1 \
UV_PYTHON_DOWNLOADS=0 \
PRISMA_USE_GLOBAL_NODE=true \
PATH="/app/.venv/bin:${PATH}"
# Stage 1 — install dependencies only.
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
--mount=type=bind,source=uv.lock,target=uv.lock \
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--python python3.13
# Stage 2 — copy source and install the project + workspace members.
COPY . .
RUN --mount=type=cache,target=/root/.cache/uv \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--python python3.13
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh
# ---------- Runtime ----------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
RUN for i in 1 2 3; do \
apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with
# /home/nonroot. We run the backend as that user
WORKDIR /app
ENV HOME=/home/nonroot \
PATH="/app/.venv/bin:${PATH}" \
PYTHONPATH="/app" \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1 \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
COPY --from=builder --chown=nonroot:nonroot /app /app
COPY --from=builder /opt/prisma /opt/prisma
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
find /app/.venv -type d -path "*/tornado/test" -delete && \
chmod -R a+rX /opt/prisma && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
USER nonroot
EXPOSE 4001/tcp
ENTRYPOINT ["/app/docker/component_entrypoint.sh", "uvicorn", "backend.main:app"]
CMD ["--host", "0.0.0.0", "--port", "4001"]

View file

@ -4,6 +4,5 @@
- Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM)
- Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM)
- Grype scan on `Dockerfile.database` (fails on CRITICAL)
- Grype scan on main `Dockerfile` (fails on CRITICAL)
- Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0)

View file

@ -1,25 +0,0 @@
FROM ollama/ollama as ollama
RUN echo "auto installing llama2"
# auto install ollama/llama2
RUN ollama serve & sleep 2 && ollama pull llama2
RUN echo "installing litellm"
RUN apt-get update
# Install Python
RUN apt-get install -y python3 python3-pip
# Set the working directory in the container
WORKDIR /app
# Copy the current directory contents into the container at /app
COPY . /app
# Install any needed packages specified in requirements.txt
RUN python3 -m pip install litellm
COPY start.sh /start.sh
ENTRYPOINT [ "/bin/bash", "/start.sh" ]

View file

@ -1 +0,0 @@
litellm==1.83.14

View file

@ -1,2 +0,0 @@
ollama serve &
litellm

View file

@ -1,35 +0,0 @@
import openai
api_base = "http://0.0.0.0:8000"
openai.api_base = api_base
openai.api_key = "temp-key"
print(openai.api_base)
print("LiteLLM: response from proxy with streaming")
response = openai.ChatCompletion.create(
model="ollama/llama2",
messages=[
{
"role": "user",
"content": "this is a test request, acknowledge that you got it",
}
],
stream=True,
)
for chunk in response:
print(f"LiteLLM: streaming response from proxy {chunk}")
response = openai.ChatCompletion.create(
model="ollama/llama2",
messages=[
{
"role": "user",
"content": "this is a test request, acknowledge that you got it",
}
],
)
print(f"LiteLLM: response from proxy {response}")

View file

@ -1,46 +0,0 @@
services:
# Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints.
# Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage.
litellm:
build:
context: .
dockerfile: docker/Dockerfile.non_root
target: runtime
args:
PROXY_EXTRAS_SOURCE: "local"
depends_on:
- squid
user: "101:101"
group_add:
- "2345"
read_only: true
cap_drop:
- ALL
security_opt:
- no-new-privileges:true
tmpfs:
- /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777
- /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777
volumes:
- ./proxy_server_config.yaml:/app/config.yaml:ro
environment:
LITELLM_NON_ROOT: "true"
PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries"
XDG_CACHE_HOME: "/app/cache"
LITELLM_MIGRATION_DIR: "/app/migrations"
HTTP_PROXY: "http://squid:3128"
HTTPS_PROXY: "http://squid:3128"
NO_PROXY: "localhost,127.0.0.1,db"
command:
- "--port"
- "4000"
- "--config"
- "/app/config.yaml"
squid:
image: sameersbn/squid:3.5.27-2
restart: unless-stopped
ports:
- "3128:3128"
tmpfs:
- /var/spool/squid:rw,noexec,nosuid,nodev,size=64m
- /var/log/squid:rw,noexec,nosuid,nodev,size=16m

View file

@ -1,52 +1,69 @@
# LiteLLM plus a Postgres database that stores models, virtual keys and spend
# logs. Used by https://docs.litellm.ai/docs/proxy/docker_quick_start
#
# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml
# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
# docker compose up -d
#
# `docker compose up` runs the published image. From a checkout of the repo,
# `docker compose up --build` builds the Dockerfile next to this file instead.
# `docker compose --profile monitoring up -d` also starts Prometheus on 9090.
#
# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY
# makes credentials already stored in the database unreadable. For anything
# beyond local evaluation, pin the image to a specific release tag.
services:
litellm:
image: docker.litellm.ai/berriai/litellm:main-stable
build:
context: .
args:
target: runtime
image: docker.litellm.ai/berriai/litellm:main-stable
#########################################
## Uncomment these lines to start proxy with a config.yaml file ##
# volumes:
# - ./config.yaml:/app/config.yaml
# command:
# - "--config=/app/config.yaml"
##############################################
target: runtime
pull_policy: missing
ports:
- "4000:4000" # Map the container port to the host, change the host port if necessary
- "4000:4000"
# The same image runs every component. The first word of `command` picks
# it: proxy (default), gateway, backend, ui, migrations, metrics,
# collector. See docker/README.md
# command: ["--config", "/app/config.yaml"]
# volumes:
# - ./config.yaml:/app/config.yaml:ro
environment:
DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm"
# Optional: route read-only queries (find_*, count, group_by, query_raw/_first)
# to a separate reader endpoint, e.g. an Aurora reader. Leave unset for
# single-DB deployments. With IAM_TOKEN_DB_AUTH enabled, the reader URL
# is auto-refreshed alongside the writer.
# DATABASE_URL_READ_REPLICA: "postgresql://llmproxy:dbpassword9090@db-reader:5432/litellm"
STORE_MODEL_IN_DB: "True" # allows adding models to proxy via UI
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env, see the header of this file}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env, see the header of this file}
DATABASE_URL: postgresql://llmproxy:dbpassword9090@db:5432/litellm
STORE_MODEL_IN_DB: "True"
env_file:
- .env # Load local .env file
- path: .env
required: false
read_only: true
cap_drop:
- ALL
security_opt:
- no-new-privileges:true
tmpfs:
- /tmp:rw,noexec,nosuid,nodev,size=64m,mode=1777
- /app/.cache:rw,noexec,nosuid,nodev,size=128m,mode=1777
depends_on:
- db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
healthcheck: # Defines the health check configuration for the container
db:
condition: service_healthy
healthcheck:
test:
- CMD-SHELL
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')" # Command to execute for health check
interval: 30s # Perform health check every 30 seconds
timeout: 10s # Health check command times out after 10 seconds
retries: 3 # Retry up to 3 times if health check fails
start_period: 40s # Wait 40 seconds after container start before beginning health checks
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"
interval: 30s
timeout: 10s
retries: 3
start_period: 40s
db:
image: postgres:16
restart: always
container_name: litellm_db
environment:
POSTGRES_DB: litellm
POSTGRES_USER: llmproxy
POSTGRES_PASSWORD: dbpassword9090
ports:
- "5432:5432"
volumes:
- postgres_data:/var/lib/postgresql/data # Persists Postgres data across container restarts
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -d litellm -U llmproxy"]
interval: 1s
@ -55,6 +72,8 @@ services:
prometheus:
image: prom/prometheus
profiles:
- monitoring
volumes:
- prometheus_data:/prometheus
- ./prometheus.yml:/etc/prometheus/prometheus.yml
@ -68,6 +87,5 @@ services:
volumes:
prometheus_data:
driver: local
postgres_data:
name: litellm_postgres_data # Named volume for Postgres data persistence
name: litellm_postgres_data

93
docker-entrypoint.sh Executable file
View file

@ -0,0 +1,93 @@
#!/bin/sh
# LiteLLM image entrypoint: docker-entrypoint.sh [COMPONENT] [ARGS...]
#
# COMPONENT selects the process this container runs; it can also be given as
# LITELLM_COMPONENT when the command carries no component. Anything after it is
# passed to that process. A first argument starting with "-" (or no argument at
# all) keeps the historical behaviour of running the monolithic proxy.
#
# proxy litellm ARGS everything in one process (default)
# gateway python -m gateway.launch ARGS inference routes, 0.0.0.0:4000
# backend uvicorn backend.main:app ARGS management routes, 0.0.0.0:4001
# ui nginx serving the admin UI port 3000
# migrations python migrations/run.py prisma migrate deploy, then exit
# metrics python -m litellm.proxy.prometheus_metrics_server ARGS
# collector python -m litellm.proxy.collector ARGS
#
# gateway and backend get their host and port defaults first, so ARGS such as
# --port 8080 override them. An unknown first word is executed as-is (docker run
# <image> sh). PgBouncer is not a component: LITELLM_PGBOUNCER_ENABLED=true
# starts it inside proxy and gateway. Components that write Prometheus samples
# start with an empty PROMETHEUS_MULTIPROC_DIR; metrics, collector and raw
# commands get the directory created but keep the workers' files.
set -eu
usage() {
sed -n '2,24p' "$0" | sed 's/^# \{0,1\}//' >&2
}
component="${LITELLM_COMPONENT:-proxy}"
case "${1:-}" in
proxy|gateway|backend|ui|migrations|metrics|collector)
component="$1"
shift
;;
""|-*)
;;
help)
usage
exit 0
;;
*)
[ -z "${PROMETHEUS_MULTIPROC_DIR:-}" ] || mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
exec "$@"
;;
esac
case "$component" in
proxy)
set -- litellm "$@"
;;
gateway)
set -- python -m gateway.launch --workers "${NUM_WORKERS:-1}" --host 0.0.0.0 --port 4000 "$@"
;;
backend)
set -- uvicorn backend.main:app --host 0.0.0.0 --port 4001 "$@"
;;
ui)
exec nginx -g 'daemon off;' "$@"
;;
migrations)
set -- python /app/migrations/run.py "$@"
;;
metrics)
set -- python -m litellm.proxy.prometheus_metrics_server "$@"
;;
collector)
set -- python -m litellm.proxy.collector "$@"
;;
*)
echo "docker-entrypoint.sh: unknown LITELLM_COMPONENT '$component'" >&2
usage
exit 64
;;
esac
if [ -n "${PROMETHEUS_MULTIPROC_DIR:-}" ]; then
case "$component" in
metrics|collector) mkdir -p "$PROMETHEUS_MULTIPROC_DIR" ;;
*)
mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db
;;
esac
fi
case "${USE_DDTRACE:-}" in
[Tt][Rr][Uu][Ee])
export DD_TRACE_OPENAI_ENABLED="False"
exec ddtrace-run "$@"
;;
esac
exec "$@"

View file

@ -1,162 +0,0 @@
# syntax=docker/dockerfile:1.7
# Base image for building
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
# Runtime image
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
ARG PGBOUNCER_VERSION=1.25.2
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
FROM $UV_IMAGE AS uvbin
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
ARG PGBOUNCER_VERSION
ARG PGBOUNCER_SHA256
USER root
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
WORKDIR /build
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
tar xzf pgbouncer.tar.gz --strip-components=1 && \
./configure --prefix=/usr/local --with-openssl=/usr && \
make -j"$(nproc)" pgbouncer && \
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
# Admin UI builder. Pinned to the build platform so the architecture-independent
# Next.js static export compiles once natively even in a multi-arch build,
# instead of once per target arch under QEMU.
FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder
ENV NEXT_TELEMETRY_DISABLED=1 \
npm_config_fund=false \
npm_config_audit=false
WORKDIR /ui
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline
COPY ui/litellm-dashboard/ ./
RUN npm run build
FROM $LITELLM_BUILD_IMAGE AS builder
WORKDIR /app
USER root
COPY --from=uvbin /uv /usr/local/bin/uv
COPY --from=uvbin /uvx /usr/local/bin/uvx
RUN apk add --no-cache \
bash \
gcc \
python-3.13 \
python-3.13-dev \
openssl \
openssl-dev \
nodejs \
npm \
libsndfile
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
UV_PYTHON_DOWNLOADS=0 \
PATH="/app/.venv/bin:${PATH}"
# Copy dependency metadata first for layer caching
COPY pyproject.toml uv.lock ./
COPY enterprise/pyproject.toml enterprise/
COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/
# Install third-party dependencies (cached unless pyproject.toml/uv.lock change)
RUN uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13
# Copy full source tree
COPY . .
# Replace the committed UI bundle with the one built from this exact source.
# Clearing first drops the committed bundle's content-hashed chunks that COPY
# would otherwise leave behind alongside the fresh ones.
RUN rm -rf litellm/proxy/_experimental/out
COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/
# Build Admin UI before final sync (applies the enterprise color override when present)
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
# Install project and workspace packages (fast - deps already cached)
RUN uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
# node (without npm) is required by the prisma CLI at runtime
RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
WORKDIR /app
ENV PATH="/app/.venv/bin:${PATH}" \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
PRISMA_OFFLINE_MODE=true
# Copy only what runtime needs. The application is installed inside the venv;
# the rest of the builder's /app is source and build metadata that must not
# ship (manifest-scanning tools attribute everything in it to this image).
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
COPY --from=builder /app/.venv /app/.venv
COPY --from=builder /app/docker /app/docker
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
# enterprise/ is imported by source path at runtime (proxy_cli puts the
# working directory on sys.path; litellm/proxy/hooks resolves
# enterprise.enterprise_hooks from it)
COPY --from=builder /app/enterprise /app/enterprise
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every
# runtime uid can read and that no cache volume mount shadows (unlike
# /app/.cache or $HOME/.cache in readOnlyRootFilesystem + emptyDir setups).
# The paths are pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and
# recorded into the generated client at build time, so `prisma migrate
# deploy` on a fresh database needs no npm and no network access
# (#33650, #24554).
COPY --from=builder /opt/prisma /opt/prisma
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
find /app/.venv -type d -path "*/tornado/test" -delete && \
chmod -R a+rX /opt/prisma && \
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
EXPOSE 4000/tcp
ENTRYPOINT ["docker/prod_entrypoint.sh"]
CMD ["--port", "4000"]

View file

@ -1,217 +0,0 @@
# syntax=docker/dockerfile:1.7
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG PROXY_EXTRAS_SOURCE=published
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
ARG PGBOUNCER_VERSION=1.25.2
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
FROM $UV_IMAGE AS uvbin
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
ARG PGBOUNCER_VERSION
ARG PGBOUNCER_SHA256
USER root
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
WORKDIR /build
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
tar xzf pgbouncer.tar.gz --strip-components=1 && \
./configure --prefix=/usr/local --with-openssl=/usr && \
make -j"$(nproc)" pgbouncer && \
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
# Admin UI builder. Pinned to the build platform so the architecture-independent
# Next.js static export compiles once natively even in a multi-arch build,
# instead of once per target arch under QEMU.
FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder
ENV NEXT_TELEMETRY_DISABLED=1 \
npm_config_fund=false \
npm_config_audit=false
WORKDIR /ui
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline
COPY ui/litellm-dashboard/ ./
RUN npm run build
FROM $LITELLM_BUILD_IMAGE AS builder
ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
USER root
COPY --from=uvbin /uv /usr/local/bin/uv
COPY --from=uvbin /uvx /usr/local/bin/uvx
RUN for i in 1 2 3; do \
apk add --no-cache \
python-3.13 \
python-3.13-dev \
gcc \
rust \
bash \
coreutils \
curl \
openssl \
libsndfile \
nodejs \
npm && break || sleep 5; \
done
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
UV_PYTHON_DOWNLOADS=0 \
PATH="/app/.venv/bin:${PATH}" \
LITELLM_NON_ROOT=true \
XDG_CACHE_HOME=/app/.cache
# Copy dependency metadata first for layer caching
COPY pyproject.toml uv.lock ./
COPY enterprise/pyproject.toml enterprise/
COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/
# Install third-party dependencies (cached unless pyproject.toml/uv.lock change)
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13
# Copy full source tree
COPY . .
# Replace the committed UI bundle with the one built from this exact source.
# Clearing first drops the committed bundle's content-hashed chunks that COPY
# would otherwise leave behind alongside the fresh ones.
RUN rm -rf litellm/proxy/_experimental/out
COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/
# Set non-root flag for build time consistency
ENV LITELLM_NON_ROOT=true
RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
cp -r /app/litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \
cp /app/litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \
touch /var/lib/litellm/ui/.litellm_ui_ready
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13 \
--no-sources-package litellm-proxy-extras; \
else \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13; \
fi
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
USER root
RUN for i in 1 2 3; do \
apk upgrade --no-cache && break || sleep 5; \
done && \
for i in 1 2 3; do \
apk add --no-cache python-3.13 bash openssl tzdata libsndfile nodejs libevent && break || sleep 5; \
done
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
# Copy only what runtime needs. The application is installed inside the venv;
# the rest of the builder's /app is source and build metadata that must not
# ship (manifest-scanning tools attribute everything in it to this image).
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
COPY --from=builder /app/.venv /app/.venv
COPY --from=builder /app/docker /app/docker
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
# enterprise/ is imported by source path at runtime (proxy_cli puts the
# working directory on sys.path; litellm/proxy/hooks resolves
# enterprise.enterprise_hooks from it)
COPY --from=builder /app/enterprise /app/enterprise
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every runtime
# uid can read and that no cache volume mount shadows (unlike /app/.cache or
# $HOME/.cache under readOnlyRootFilesystem + emptyDir or arbitrary-uid setups).
# PRISMA_CLI_QUERY_ENGINE_TYPE=binary makes the CLI use the baked binary query
# engine directly, so `prisma migrate deploy` on a fresh database needs no npm
# and no network access; without it the CLI looks for the library engine, which
# prisma stopped baking, and falls back to a download that fails offline or as a
# non-writable uid (#33650, #24554).
COPY --from=builder /opt/prisma /opt/prisma
COPY --from=builder /var/lib/litellm/ui /var/lib/litellm/ui
COPY --from=builder /var/lib/litellm/assets /var/lib/litellm/assets
# XDG_CACHE_HOME is intentionally left unset so it falls back to $HOME/.cache
# (/app/.cache, writable by the runtime uid). The prisma bake at the read-only
# /opt/prisma is anchored by PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH, so
# nothing needs XDG to point there; pointing it at the read-only bake would
# deny any XDG-aware library that writes a cache at runtime.
ENV PATH="/app/.venv/bin:${PATH}" \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
HOME=/app \
LITELLM_NON_ROOT=true \
PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
PRISMA_HIDE_UPDATE_MESSAGE=1 \
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
PRISMA_OFFLINE_MODE=true
RUN mkdir -p /nonexistent /app/.cache /var/lib/litellm/assets /var/lib/litellm/ui && \
chown -R nobody:nogroup /app /var/lib/litellm/ui /var/lib/litellm/assets /nonexistent && \
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
chown -R nobody:nogroup "$PRISMA_PATH" && \
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup "$LITELLM_PKG_MIGRATIONS_PATH" || true && \
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
chgrp -R 0 "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 "$LITELLM_PROXY_EXTRAS_PATH" || true && \
chmod -R g=u "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u "$LITELLM_PROXY_EXTRAS_PATH" || true && \
chmod -R g+w "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w "$LITELLM_PROXY_EXTRAS_PATH" || true && \
chmod -R g+rX "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
chmod -R a+rX /opt/prisma && \
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null 2>&1 && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
USER 65534
EXPOSE 4000/tcp
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
CMD ["--port", "4000"]

View file

@ -2,15 +2,15 @@
This guide provides instructions for building and running the LiteLLM application using Docker and Docker Compose.
> **Just want to run LiteLLM?** This guide builds from source. To run the published
> image instead, use `docker-compose.quickstart.yml` in this directory — the
> two-service stack (gateway + Postgres) that the
> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents:
> **Just want to run LiteLLM?** `docker-compose.yml` in the repository root runs
> the published image with a Postgres database, the stack the
> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents,
> and it works on its own outside a checkout:
>
> ```bash
> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml
> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml
> printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
> docker compose -f docker-compose.quickstart.yml up -d
> docker compose up -d
> ```
## Prerequisites
@ -20,33 +20,52 @@ This guide provides instructions for building and running the LiteLLM applicatio
## Building and Running the Application
To build and run the application, you will use the `docker-compose.yml` file located in the root of the project. This file is configured to use the `Dockerfile.non_root` for a secure, non-root container environment.
The same `docker-compose.yml` builds from source when you pass `--build`: it builds the `Dockerfile` in the repository root, the one image LiteLLM ships. `docker-entrypoint.sh` next to it is the only entrypoint script; every container starts through it
### 1. Set the Master Key
## One image, many components
The application requires a `LITELLM_MASTER_KEY` for signing and validating tokens. You must set this key as an environment variable before running the application.
Every LiteLLM container runs the same image. The first word of the container command (or the `LITELLM_COMPONENT` environment variable when the command carries only flags) picks the process the container runs, and everything after it is handed to that process unchanged:
Create a `.env` file in the root of the project and add the following line:
| Component | Runs | Port |
|--------------|-----------------------------------------------------|------|
| `proxy` | `litellm ...` (everything in one process, default) | 4000 |
| `gateway` | `python -m gateway.launch ...` (inference routes) | 4000 |
| `backend` | `uvicorn backend.main:app ...` (management routes) | 4001 |
| `ui` | nginx serving the static admin UI | 3000 |
| `migrations` | `python migrations/run.py`, `prisma migrate deploy` then exit | |
| `metrics` | `python -m litellm.proxy.prometheus_metrics_server ...` | `--port` |
| `collector` | `python -m litellm.proxy.collector ...` | |
```
LITELLM_MASTER_KEY=your-secret-key
```bash
docker run -p 4000:4000 litellm --config /app/config.yaml # proxy, exactly as before
docker run -p 4000:4000 litellm gateway --port 4000 # componentized data plane
docker run -p 4001:4001 -e LITELLM_COMPONENT=backend litellm # same, chosen through the env
docker run -p 3000:3000 --read-only --tmpfs /tmp litellm ui # admin UI behind nginx
docker run -e DATABASE_URL=... litellm migrations # one-off schema migration job
docker run -it litellm sh # anything else runs verbatim
```
Replace `your-secret-key` with a strong, randomly generated secret.
PgBouncer is not a separate component: `LITELLM_PGBOUNCER_ENABLED=true` starts an in-container PgBouncer in front of `DATABASE_URL` inside `proxy` and `gateway`. `USE_DDTRACE=true` wraps whichever component runs with `ddtrace-run`, and `PROMETHEUS_MULTIPROC_DIR` is emptied of stale samples before any workers fork (the `metrics` and `collector` sidecars only read it, so their restart keeps the live samples)
The image runs as uid `65532` (`nonroot` in the Wolfi base) and also works as an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, because everything it writes at runtime lives under `/app/.cache`, `/var/lib/litellm` and `/tmp`. Mount those (or set `readOnlyRootFilesystem` with emptyDirs there) for a read-only root filesystem. Prisma's CLI and engines are baked under `/opt/prisma`, so migrations need neither network nor a writable home
### 1. Set the Master and Salt Keys
The proxy signs virtual keys with `LITELLM_MASTER_KEY` and encrypts stored provider credentials with `LITELLM_SALT_KEY`. Compose reads both from a `.env` file in the directory you run it from, so generate them once:
```bash
printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
```
Keep the file: regenerating `LITELLM_SALT_KEY` makes credentials already stored in the database unreadable. Provider keys such as `OPENAI_API_KEY` go in the same file, and the whole file is passed to the container
### 2. Build and Run the Containers
Once you have set the `LITELLM_MASTER_KEY`, you can build and run the containers using the following command:
```bash
docker compose up -d --build
```
This command will:
- Build the Docker image using `Dockerfile.non_root`.
- Start the `litellm`, `litellm_db`, and `prometheus` services in detached mode (`-d`).
- The `--build` flag ensures that the image is rebuilt if there are any changes to the Dockerfile or the application code.
This command builds the image from the root `Dockerfile` and starts the `litellm` and `db` services in detached mode. Without `--build`, `docker compose up` pulls the published `main-stable` image instead. Add `--profile monitoring` to also start Prometheus on port 9090, scraping the proxy with the root `prometheus.yml`
### 3. Verifying the Application is Running
@ -70,34 +89,18 @@ To stop the running containers, use the following command:
docker compose down
```
## Hardened / Offline Testing
## Hardening
To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file:
The compose file runs the proxy the way a locked-down cluster would: as the image's non-root user with a read-only root filesystem, every capability dropped, `no-new-privileges` set, and tmpfs mounts only at `/tmp` and `/app/.cache`. The image is built for that, so a change that makes the proxy write anywhere else fails here before it fails in Kubernetes. To try an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, add `user: "101:0"` to the `litellm` service
Prisma's CLI and engines are baked under `/opt/prisma`, so migrations run without network access. Verify with:
```bash
docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache
docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d
docker run --rm --network none --entrypoint prisma docker.litellm.ai/berriai/litellm:main-stable --version
```
This setup:
- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image.
- Runs the proxy as a non-root user with a read-only rootfs and only writable tmpfs mounts:
- `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`)
- `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`)
- Pre-builds and serves the admin UI from read-only paths:
- `/var/lib/litellm/ui` (pre-restructured Next.js UI with `.litellm_ui_ready` marker)
- `/var/lib/litellm/assets` (UI logos and assets)
- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines.
You should also verify offline Prisma behaviour with:
```bash
docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version
```
This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access.
## Troubleshooting
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
- **`Master key is not initialized`**: This error means the `LITELLM_MASTER_KEY` environment variable is not set. Make sure you have created a `.env` file in the project root with the `LITELLM_MASTER_KEY` defined.
- **`required variable LITELLM_MASTER_KEY is missing a value`**: Compose did not find a `.env` file with `LITELLM_MASTER_KEY` and `LITELLM_SALT_KEY` in the directory you ran it from. Generate one as shown above.
- **`password authentication failed for user "llmproxy"`**: the `litellm_postgres_data` volume was initialised by an older compose file with different credentials. `docker compose down -v` drops it and the next `up` recreates the database.
- **`build_admin_ui.sh: not found`**: the build context is wrong. Run `docker compose` from the root of the repository.

View file

@ -1,64 +0,0 @@
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.10.9@sha256:10902f58a1606787602f303954cea099626a4adb02acbac4c69920fe9d278f82
FROM $UV_IMAGE AS uvbin
FROM python:3.13-slim@sha256:739e7213785e88c0f702dcdc12c0973afcbd606dbf021a589cab77d6b00b579d
ARG LITELLM_VERSION=1.83.0
WORKDIR /app
COPY --from=uvbin /uv /usr/local/bin/uv
COPY --from=uvbin /uvx /usr/local/bin/uvx
RUN apt-get update && \
apt-get install -y --no-install-recommends gcc libffi-dev nodejs npm && \
rm -rf /var/lib/apt/lists/*
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
PATH="/app/.venv/bin:${PATH}"
COPY schema.prisma .
# This image is specifically for validating/installing the published PyPI
# artifact, not the checked-out source tree.
# Keep the moved proxy-runtime packages explicit until the published PyPI
# artifact includes that extra; newer releases will simply dedupe these.
RUN uv venv --python python && \
uv pip install --python /app/.venv/bin/python \
"litellm[proxy,proxy-runtime]==${LITELLM_VERSION}" \
"google-cloud-aiplatform==1.133.0" \
"google-genai==1.37.0" \
"anthropic[vertex]==0.84.0" \
"grpcio==1.78.0" \
"prometheus-client==0.20.0" \
"langfuse==2.59.7" \
"opentelemetry-api==1.28.0" \
"opentelemetry-sdk==1.28.0" \
"opentelemetry-exporter-otlp==1.28.0" \
"ddtrace==4.11.0" \
"sentry-sdk==2.21.0" \
"mangum==0.17.0" \
"azure-ai-contentsafety==1.0.0" \
"azure-storage-file-datalake==12.20.0" \
"pypdf==6.7.5" \
"llm-sandbox==0.3.31" \
"detect-secrets==1.5.0" \
"prisma==0.11.0" \
"openai==2.24.0"
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma && \
chmod -R a+rX /opt/prisma && \
python -c "import sys; from prisma.client import BINARY_PATHS; bad = sorted(p for group in BINARY_PATHS.model_dump().values() for p in group.values() if not p.startswith('/opt/prisma/')); sys.exit('prisma engines baked outside /opt/prisma: %r' % bad) if bad else None"
ENV PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
COPY docker/prod_entrypoint.sh /app/docker/prod_entrypoint.sh
RUN sed -i 's/\r$//' /app/docker/prod_entrypoint.sh && chmod +x /app/docker/prod_entrypoint.sh
EXPOSE 4000/tcp
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
CMD ["--port", "4000"]

View file

@ -1,9 +0,0 @@
# Docker to build LiteLLM Proxy from litellm pip package
### When to use this ?
If you need to build LiteLLM Proxy from litellm pip package, you can use this Dockerfile as a reference.
### Why build from pip package ?
- If your company has a strict requirement around security / building images you can follow steps outlined here

View file

@ -1,16 +0,0 @@
#!/bin/sh
# stale samples from a previous container incarnation would be summed into the aggregate
if [ -n "$PROMETHEUS_MULTIPROC_DIR" ]; then
mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db
fi
case "$USE_DDTRACE" in
[Tt][Rr][Uu][Ee])
export DD_TRACE_OPENAI_ENABLED="False"
exec ddtrace-run "$@"
;;
esac
exec "$@"

View file

@ -1,41 +0,0 @@
# LiteLLM quickstart stack: the gateway plus a Postgres database that stores
# models, virtual keys, and spend logs. Used by
# https://docs.litellm.ai/docs/proxy/docker_quick_start
#
# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml
# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
# docker compose -f docker-compose.quickstart.yml up -d
#
# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY
# makes credentials already stored in the database unreadable. For anything
# beyond local evaluation, pin the image to a specific release tag.
services:
litellm:
image: docker.litellm.ai/berriai/litellm:main-stable
ports:
- "4000:4000"
environment:
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file}
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
STORE_MODEL_IN_DB: "True"
depends_on:
db:
condition: service_healthy
db:
image: postgres:16
environment:
POSTGRES_USER: litellm
POSTGRES_PASSWORD: litellm
POSTGRES_DB: litellm
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm"]
interval: 5s
timeout: 5s
retries: 10
volumes:
- postgres_data:/var/lib/postgresql/data
volumes:
postgres_data:

View file

@ -1,16 +0,0 @@
#!/bin/bash
set -euo pipefail
REPO_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
VENV_PYTHON="$REPO_ROOT/.venv/bin/python"
MIGRATION_SCRIPT="$REPO_ROOT/litellm/proxy/prisma_migration.py"
if [ -x "$VENV_PYTHON" ]; then
"$VENV_PYTHON" "$MIGRATION_SCRIPT"
elif command -v uv >/dev/null 2>&1; then
(cd "$REPO_ROOT" && uv run --no-sync python "$MIGRATION_SCRIPT")
else
python3 "$MIGRATION_SCRIPT"
fi
echo "Migration script ran successfully!"

View file

@ -1,4 +0,0 @@
#!/bin/bash
set -euo pipefail
# semantic-router dependencies are installed via `uv sync`.

View file

@ -1,10 +0,0 @@
#!/bin/sh
case "$USE_DDTRACE" in
[Tt][Rr][Uu][Ee])
export DD_TRACE_OPENAI_ENABLED="False"
exec ddtrace-run litellm "$@"
;;
esac
exec litellm "$@"

View file

@ -1,18 +0,0 @@
schemaVersion: 2.0.0
metadataTest:
entrypoint: ["docker/prod_entrypoint.sh"]
user: "65534"
workdir: "/app"
fileExistenceTests:
- name: "Prisma Folder"
path: "/usr/local/lib/python3.13/site-packages/prisma/"
shouldExist: true
uid: 65534
gid: 65534
- name: "Prisma Schema"
path: "/usr/local/lib/python3.13/site-packages/prisma/schema.prisma"
shouldExist: true
uid: 65534
gid: 65534

View file

@ -1,126 +0,0 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
ARG PGBOUNCER_VERSION=1.25.2
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
FROM $UV_IMAGE AS uvbin
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
ARG PGBOUNCER_VERSION
ARG PGBOUNCER_SHA256
USER root
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
WORKDIR /build
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
tar xzf pgbouncer.tar.gz --strip-components=1 && \
./configure --prefix=/usr/local --with-openssl=/usr && \
make -j"$(nproc)" pgbouncer && \
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
# ---------- Builder ----------
FROM $LITELLM_BUILD_IMAGE AS builder
WORKDIR /app
USER root
COPY --from=uvbin /uv /uvx /usr/local/bin/
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
RUN for i in 1 2 3; do \
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start.
# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a
# BuildKit cache mount (different filesystem).
# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of
# silently pulling a managed interpreter.
# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't
# silently re-enable nodeenv's Node download.
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
UV_COMPILE_BYTECODE=1 \
UV_PYTHON_DOWNLOADS=0 \
PRISMA_USE_GLOBAL_NODE=true \
PATH="/app/.venv/bin:${PATH}"
# Stage 1 — install dependencies only.
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
--mount=type=bind,source=uv.lock,target=uv.lock \
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra bedrock-realtime \
--python python3.13
# Stage 2 — copy source and install the project + workspace members.
COPY . .
RUN --mount=type=cache,target=/root/.cache/uv \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra bedrock-realtime \
--python python3.13
# PYTHONPATH=/app makes the source tree shadow the installed package, so the
# compiled Rust extension must live next to the source or it is never imported.
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh
# ---------- Runtime ----------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
RUN for i in 1 2 3; do \
apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic libevent && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with
# /home/nonroot. We run the proxy as that user.
WORKDIR /app
ENV HOME=/home/nonroot \
PATH="/app/.venv/bin:${PATH}" \
PYTHONPATH="/app" \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1 \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
COPY --from=builder --chown=nonroot:nonroot /app /app
COPY --from=builder /opt/prisma /opt/prisma
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
find /app/.venv -type d -path "*/tornado/test" -delete && \
chmod -R a+rX /opt/prisma && \
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \
python -c "import litellm; from litellm.rust_bridge.loader import native_bridge_available; assert litellm.__file__ == '/app/litellm/__init__.py', litellm.__file__; assert native_bridge_available()"
USER nonroot
EXPOSE 4000/tcp
ENTRYPOINT ["sh", "-c", "exec /app/docker/component_entrypoint.sh python -m gateway.launch --workers \"${NUM_WORKERS:-1}\" \"$@\"", "--"]
CMD ["--host", "0.0.0.0", "--port", "4000"]

View file

@ -1,2 +0,0 @@
#!/bin/bash
python3 proxy_cli.py

View file

@ -1,120 +0,0 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin
# ---------- Builder ----------
#
# Minimal install for `prisma migrate deploy`. We deliberately skip the heavy
# `proxy-runtime` (otel, sentry, ddtrace, pypdf, google-genai, anthropic-vertex,
# ...) and `semantic-router` extras that the gateway/backend pull in — the
# migration engine doesn't need them. We DO install `--extra proxy` so the
# DB-URL helper from `litellm.proxy.auth.rds_iam_token` is importable, which
# is how the gateway and backend assemble `DATABASE_URL` at pod startup when
# `IAM_TOKEN_DB_AUTH=true` (see backend/main.py:17, gateway/main.py:22). And
# `--extra extra_proxy` provides the `prisma` CLI + the secret-manager
# backends `litellm.secret_managers.main` lazily imports.
#
# `prisma generate` runs once at BUILD time to (a) install the Node-based
# Prisma CLI into the binary cache and (b) download the migration / query
# engine binaries. The Python client it also produces is unused by this
# image's runtime entrypoint — that's fine, it's a few hundred KB and the
# alternative (`prisma py fetch`) doesn't reliably trigger engine downloads
# under nodeenv. Crucially we do NOT run `prisma generate` at RUNTIME; the
# old migration job did, on every pod start, which is the wasteful behaviour
# the componentization is fixing.
FROM $LITELLM_BUILD_IMAGE AS builder
WORKDIR /app
USER root
COPY --from=uvbin /uv /uvx /usr/local/bin/
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
RUN for i in 1 2 3; do \
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
UV_LINK_MODE=copy \
UV_COMPILE_BYTECODE=1 \
UV_PYTHON_DOWNLOADS=0 \
PRISMA_USE_GLOBAL_NODE=true \
PATH="/app/.venv/bin:${PATH}"
# Stage 1 — install third-party deps only (cached by pyproject.toml/uv.lock).
RUN --mount=type=cache,target=/root/.cache/uv \
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
--mount=type=bind,source=uv.lock,target=uv.lock \
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
--extra proxy \
--extra extra_proxy \
--python python3.13
# Stage 2 — copy source and install the project + workspace members.
COPY . .
RUN --mount=type=cache,target=/root/.cache/uv \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra extra_proxy \
--python python3.13
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
COPY migrations/run.py /app/run.py
# Pre-warm the Prisma binary cache so the Job pod doesn't reach the
# internet on first start. This matches what the backend Dockerfile does:
# `prisma generate` runs nodeenv (downloads Node), installs the prisma npm
# CLI, downloads the engine binaries for each `binaryTarget` in
# schema.prisma, AND emits the generated Python client. We don't need the
# client at runtime — the migration job invokes `prisma migrate deploy`
# via subprocess — but having it cached is harmless and the alternative
# (`prisma py fetch`) doesn't reliably trigger engine downloads.
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
prisma generate --schema=./schema.prisma
# ---------- Runtime ----------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
RUN for i in 1 2 3; do \
apk add --no-cache bash openssl tzdata python-3.13 nodejs libsndfile libatomic && break; \
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
sleep 5; \
done
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The
# Prisma engine binaries are dynamically linked against libssl/libcrypto, so
# openssl stays in the runtime layer.
WORKDIR /app
ENV HOME=/home/nonroot \
PATH="/app/.venv/bin:${PATH}" \
PYTHONPATH="/app" \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1 \
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
PRISMA_OFFLINE_MODE=true
COPY --from=builder --chown=nonroot:nonroot /app /app
COPY --from=builder /opt/prisma /opt/prisma
RUN chmod -R a+rX /opt/prisma && \
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js
USER nonroot
ENTRYPOINT ["python3", "/app/run.py"]

View file

@ -5,5 +5,5 @@ model_list:
api_key: fake-key
api_base: os.environ/FAKE_OPENAI_API_BASE
general_settings:
alerting: ["slack"]
general_settings:
alerting: ["slack"]

View file

@ -1,34 +0,0 @@
FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
ARG GH_VERSION=2.101.0
ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8
ARG UV_VERSION=0.10.9
ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \
&& rm -rf /var/lib/apt/lists/*
RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \
&& echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \
&& rm /tmp/gh.tar.gz
RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \
&& echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \
&& rm /tmp/uv.tar.gz
RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator
ENV HOME=/home/populator \
LITELLM_REPO=/opt/litellm \
DISABLE_AUTOUPDATER=1
COPY --chown=populator:populator . /opt/litellm/tests/e2e/
USER populator
WORKDIR /home/populator
CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"]

View file

@ -1,12 +1,12 @@
# Render cron job for the Claude Code compatibility-matrix populator
The populator runs daily as the Render cron job `litellm-compat-matrix`
(Docker runtime, built from the `Dockerfile` in this directory) rather
(Docker runtime; the image recipe is in [Image](#image) below) rather
than as a GitHub Action or on a dedicated VM. Trade-offs:
- ✅ No machine to keep on or patch. Render builds the image from this
directory on every push to `main` that touches `tests/e2e/**` and
runs it on the schedule.
- ✅ No machine to keep on or patch. Render builds the image on every
push to `main` that touches `tests/e2e/**` and runs it on the
schedule.
- ✅ Credentials live in Render env vars and secret files, scoped to
this one service, instead of on a VM filesystem.
- ✅ The publish token still uses the `mateo-berri` account, which is a
@ -25,7 +25,6 @@ than as a GitHub Action or on a dedicated VM. Trade-offs:
| File | Purpose |
| --- | --- |
| `Dockerfile` | The image Render builds: Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the non-root user `populator` (uid/gid 1000, which is what Render's secret files are readable by). |
| `run_daily.sh` | The actual cron job. Resolves versions, clones the worktree, installs the Claude Code CLI under test, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. |
| `install_claude_code.sh` | Downloads one Claude Code release (`<version> <dest-dir>`) from the vendor's native release channel, verifies it against the sha256 in that release's `manifest.json`, and refuses a binary whose `--version` disagrees. Run by the cron and by the `compat-matrix-image` GitHub workflow. |
| `build_matrix.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.build_from_paths`. Exists only because the bash script needs *some* way to render the per-cell aggregation, and the builder is already Python. |
@ -106,7 +105,7 @@ the same values if it ever has to be rebuilt.
| Workspace | Litellm (the one that already builds the other litellm services) |
| Type | Cron job, Docker runtime |
| Repo / branch | `BerriAI/litellm` @ `main` |
| Dockerfile path | `tests/e2e/claude_code/cron_vm/Dockerfile` |
| Image | Built from the recipe under [Image](#image). Render only builds a Dockerfile that lives in the connected repo, and this repo ships exactly one Dockerfile (the LiteLLM image in its root), so the service has to point at a copy of the recipe kept with the service or at a prebuilt image |
| Docker build context | `tests/e2e` (the repo root `.dockerignore` excludes `tests`, so the context has to start below it) |
| Build filter | included paths `tests/e2e/**` |
| Schedule | `0 6 * * *` (06:00 UTC daily) |
@ -117,7 +116,7 @@ the same values if it ever has to be rebuilt.
Render mounts secret files at `/etc/secrets/<name>`, which is where
`CREDENTIALS_DIRECTORY` and `GOOGLE_APPLICATION_CREDENTIALS` in the env
example point. Render also passes env vars to `docker build` as build
args, which is why the `Dockerfile` declares no `ARG` that could ever
args, which is why the image recipe declares no `ARG` that could ever
be given a secret's name.
Creating it through the API looks like this (fill `envVars` and
@ -145,13 +144,59 @@ curl -fsS https://api.render.com/v1/services \
"plan": "4c-16g",
"region": "oregon",
"envSpecificDetails": {
"dockerfilePath": "tests/e2e/claude_code/cron_vm/Dockerfile",
"dockerfilePath": "<path to the recipe below in the repo the service builds from>",
"dockerContext": "tests/e2e"
}
}
}'
```
## Image
Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with
this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the
non-root user `populator` (uid/gid 1000, which is what Render's secret
files are readable by). This is the recipe the Render service builds;
it lives with the service because the repo ships exactly one Dockerfile,
the LiteLLM image in its root
```dockerfile
FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
ARG GH_VERSION=2.101.0
ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8
ARG UV_VERSION=0.10.9
ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \
&& rm -rf /var/lib/apt/lists/*
RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \
&& echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \
&& rm /tmp/gh.tar.gz
RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \
&& echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \
&& tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \
&& rm /tmp/uv.tar.gz
RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator
ENV HOME=/home/populator \
LITELLM_REPO=/opt/litellm \
DISABLE_AUTOUPDATER=1
COPY --chown=populator:populator . /opt/litellm/tests/e2e/
USER populator
WORKDIR /home/populator
CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"]
```
## Operating it
```bash
@ -185,7 +230,7 @@ curl -fsS "https://api.render.com/v1/services/${CRON_ID}/deploys?limit=1" \
# Build and run the image locally (docker on Apple silicon needs the
# platform flag; the context is tests/e2e, see the table above).
docker build --platform linux/amd64 \
-f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix tests/e2e
-f /path/to/compat-matrix.Dockerfile -t compat-matrix tests/e2e
docker run --rm --platform linux/amd64 \
--env-file litellm-compat-matrix.env -e SKIP_PUBLISH=1 \
-v "$PWD/secrets:/etc/secrets:ro" compat-matrix
@ -234,8 +279,8 @@ docker run --rm --platform linux/amd64 \
A CLI release that breaks a cell shows up as a green→red flip, which
withholds auto-merge on that day's docs PR for review. To rerun the
matrix on one specific CLI, set `CLAUDE_CODE_VERSION` on the run.
`gh` and `uv` stay pinned in the `Dockerfile`; bump them in a PR with
the checksum from the release's `gh_<version>_checksums.txt` and the
`gh` and `uv` stay pinned in the image recipe; bump them with the
checksum from the release's `gh_<version>_checksums.txt` and the
tarball's `.sha256` sidecar respectively.
- **A local build on Apple silicon only proves the image assembles.**
Under QEMU the Claude Code binary (a Bun executable) dies with

View file

@ -1,8 +1,8 @@
#!/usr/bin/env bash
# Daily Claude Code compatibility-matrix populator.
#
# Runs daily as the Render cron job `litellm-compat-matrix`, built from
# the Dockerfile in this directory (see README.md). The flow is:
# Runs daily as the Render cron job `litellm-compat-matrix` (see
# README.md for the image it runs in). The flow is:
#
# 1. Resolve the latest LiteLLM final release tag from the GitHub
# Releases API.

View file

@ -1,4 +1,4 @@
"""Image-level regression net for the prisma bake in the componentized images.
"""Image-level regression net for the prisma bake in the `gateway` and `backend` components.
The gateway and backend serve requests; they never shell out to the Prisma CLI
(``PrismaManager.setup_database`` is reachable only from ``proxy_cli.py``, which
@ -35,7 +35,8 @@ import pytest
IMAGE = os.getenv("LITELLM_IMAGE")
POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine")
CURL_IMAGE = os.getenv("LITELLM_TEST_CURL_IMAGE", "curlimages/curl:8.11.1")
COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", "4000")
COMPONENT = os.getenv("LITELLM_COMPONENT", "gateway")
COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", {"gateway": "4000", "backend": "4001"}[COMPONENT])
NON_ROOT_UID = "12345:0"
STARTUP_TIMEOUT_SECONDS = int(os.getenv("LITELLM_COMPONENT_STARTUP_TIMEOUT", "180"))
@ -63,7 +64,7 @@ def offline_stack():
The container runs with DISABLE_SCHEMA_UPDATE, since applying the schema is
the migration job's responsibility in this topology and needs the Prisma CLI
these images deliberately omit, and with LITELLM_LOCAL_MODEL_COST_MAP, or the
these components never invoke, and with LITELLM_LOCAL_MODEL_COST_MAP, or the
proxy spends the whole startup budget timing out on a cost-map fetch over the
network it does not have.
@ -92,7 +93,7 @@ def offline_stack():
"-e", "LITELLM_MASTER_KEY=sk-component-serve-test",
"-e", "DISABLE_SCHEMA_UPDATE=true",
"-e", "LITELLM_LOCAL_MODEL_COST_MAP=True",
IMAGE,
IMAGE, COMPONENT,
)
yield network, component
finally:

View file

@ -1,6 +1,6 @@
"""Image-level regression net for the prisma bake in the shipped runtime image.
Boots a built image's migration entrypoint the way an OpenShift / air-gapped
Boots a built image's `migrations` component the way an OpenShift / air-gapped
deployment does (an internal-only network with no egress, an arbitrary non-root
uid in GID 0) against a brand-new Postgres, and asserts the schema was created.
@ -14,6 +14,7 @@ the normal unit-test run and exercised only where an image has been built (the
image-scan workflow). Requires a working docker CLI.
"""
import shlex
import shutil
import subprocess
import uuid
@ -26,10 +27,7 @@ POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine")
MIN_TABLES = int(os.getenv("LITELLM_TEST_MIN_TABLES", "20"))
NON_ROOT_UID = "12345:0" # arbitrary uid in GID 0, as OpenShift restricted-v2 assigns
MIGRATION_INTERPRETER = os.getenv("LITELLM_MIGRATION_INTERPRETER", "python")
MIGRATION_SCRIPT = os.getenv(
"LITELLM_MIGRATION_SCRIPT", "litellm/proxy/prisma_migration.py"
)
MIGRATION_ARGS = tuple(shlex.split(os.getenv("LITELLM_MIGRATION_ARGS", "migrations")))
pytestmark = [
pytest.mark.skipif(IMAGE is None, reason="requires a built image (set LITELLM_IMAGE)"),
@ -113,8 +111,7 @@ def test_migration_offline_as_non_root_uid(offline_postgres):
"-e", f"DATABASE_URL=postgresql://postgres:pw@{pg}:5432/litellm",
"-e", "LITELLM_MASTER_KEY=sk-offline-migration-test",
"-e", "DISABLE_SCHEMA_UPDATE=false",
"-w", "/app", "--entrypoint", MIGRATION_INTERPRETER,
IMAGE, MIGRATION_SCRIPT,
IMAGE, *MIGRATION_ARGS,
check=False,
)
tables = _table_count(pg)

View file

@ -1,7 +1,7 @@
"""Image-level regression net for arbitrary-uid boot of the UI image.
"""Image-level regression net for arbitrary-uid boot of the `ui` component.
OpenShift ``restricted-v2`` ignores the image ``USER`` and assigns an
arbitrary uid in GID 0. The stock nginx base expects to start as root, so
arbitrary uid in GID 0. A stock nginx install expects to start as root, so
its cache (``/var/cache/nginx``) and pid (``/run``) paths are root-owned
755 and the master process dies at startup with
``mkdir() "/var/cache/nginx/client_temp" failed (13: Permission denied)``.
@ -63,7 +63,7 @@ def ui_container() -> Iterator[tuple[str, str]]:
"run", "-d", "--name", container, "--network", network,
"--user", ARBITRARY_UID,
"--read-only", "--tmpfs", "/tmp",
IMAGE,
IMAGE, "ui",
)
yield network, container
finally:

View file

@ -1,5 +1,5 @@
"""Unit tests for `docker/component_entrypoint.sh` and its wiring into the
componentized `gateway` / `backend` images and Terraform deployments."""
"""Unit tests for `docker-entrypoint.sh`, the component dispatcher every shipped
container runs, and for the Terraform launch commands that must agree with it."""
import json
import os
@ -11,15 +11,13 @@ from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parents[2]
COMPONENT_ENTRYPOINT = REPO_ROOT / "docker" / "component_entrypoint.sh"
PROD_ENTRYPOINT = REPO_ROOT / "docker" / "prod_entrypoint.sh"
GATEWAY_DOCKERFILE = REPO_ROOT / "gateway" / "Dockerfile"
BACKEND_DOCKERFILE = REPO_ROOT / "backend" / "Dockerfile"
BUILD_FROM_PIP_DOCKERFILE = REPO_ROOT / "docker" / "build_from_pip" / "Dockerfile.build_from_pip"
DOCKER_ENTRYPOINT = REPO_ROOT / "docker-entrypoint.sh"
DOCKERFILE = REPO_ROOT / "Dockerfile"
TERRAFORM_ECS = REPO_ROOT / "terraform" / "litellm" / "aws" / "ecs.tf"
TERRAFORM_CLOUDRUN = REPO_ROOT / "terraform" / "litellm" / "gcp" / "cloudrun.tf"
IMAGE_ENTRYPOINT_PATH = "/app/docker/component_entrypoint.sh"
IMAGE_ENTRYPOINT_PATH = "/app/docker-entrypoint.sh"
STUBBED_EXECUTABLES = ("ddtrace-run", "uvicorn", "python", "litellm", "nginx")
TRUTHY_USE_DDTRACE = ("true", "True", "TRUE", "tRuE")
FALSY_USE_DDTRACE = (None, "", "false", "False", "1", "yes", "on", "truex")
@ -37,7 +35,6 @@ _STUB_TEMPLATE = """#!/bin/sh
_ENTRYPOINT_RE = re.compile(r"^ENTRYPOINT\s+(\[.*\])\s*$", re.MULTILINE)
_CMD_RE = re.compile(r"^CMD\s+(\[.*\])\s*$", re.MULTILINE)
_COPY_RE = re.compile(r"^COPY\s+(?!--from)(\S+)\s+(\S+)\s*$", re.MULTILINE)
_APP_TARGET_RE = re.compile(r"(?:gateway|backend)\.main:app|gateway\.launch")
_TF_STRING_LOCAL_RE = re.compile(r'^\s*(\w+)\s*=\s*"((?:[^"\\]|\\.)*)"\s*$', re.MULTILINE)
_TF_INTERPOLATION_RE = re.compile(r"\$\{(local|var)\.(\w+)\}")
@ -58,53 +55,39 @@ def _write_stubs(bin_dir: Path, names: tuple[str, ...]) -> None:
stub.chmod(0o755)
def _run_entrypoint(
script: Path,
argv: tuple[str, ...],
use_ddtrace: str | None,
tmp_path: Path,
) -> tuple[str, ...]:
"""Run `script` with stubbed executables on PATH and return the recorded lines."""
bin_dir = tmp_path / "bin"
bin_dir.mkdir(parents=True)
_write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python", "litellm"))
record = tmp_path / "record.txt"
env = {
**os.environ,
def _entrypoint_env(bin_dir: Path, record: Path, overrides: dict[str, str | None]) -> dict[str, str]:
"""The container-like environment the entrypoint runs under, with `overrides` applied (None unsets)."""
cleared = ("USE_DDTRACE", "DD_TRACE_OPENAI_ENABLED", "LITELLM_COMPONENT", "NUM_WORKERS", *overrides)
return {
**{k: v for k, v in os.environ.items() if k not in cleared},
**{k: v for k, v in overrides.items() if v is not None},
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
"RECORD": str(record),
"PYTHONPATH": PYTHONPATH_SENTINEL,
}
env.pop("USE_DDTRACE", None)
env.pop("DD_TRACE_OPENAI_ENABLED", None)
if use_ddtrace is not None:
env["USE_DDTRACE"] = use_ddtrace
result = subprocess.run(
["sh", str(script), *argv],
env=env,
capture_output=True,
text=True,
check=False,
)
def _run_entrypoint(
argv: tuple[str, ...],
tmp_path: Path,
use_ddtrace: str | None = None,
**overrides: str | None,
) -> tuple[str, ...]:
"""Run the entrypoint with stubbed executables on PATH and return the recorded lines."""
bin_dir = tmp_path / "bin"
bin_dir.mkdir(parents=True)
_write_stubs(bin_dir, STUBBED_EXECUTABLES)
record = tmp_path / "record.txt"
env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace, **overrides})
result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT), *argv], env=env, capture_output=True, text=True, check=False)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
return tuple(record.read_text().splitlines()) if record.exists() else ()
def _run_shell_command(command: str, bin_dir: Path, record: Path, use_ddtrace: str | None) -> tuple[str, ...]:
"""Run a resolved Terraform launch command through `sh -c` and return the recorded lines."""
env = {
**os.environ,
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
"RECORD": str(record),
"PYTHONPATH": PYTHONPATH_SENTINEL,
}
env.pop("USE_DDTRACE", None)
env.pop("DD_TRACE_OPENAI_ENABLED", None)
if use_ddtrace is not None:
env["USE_DDTRACE"] = use_ddtrace
env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace})
result = subprocess.run(["sh", "-c", command], env=env, capture_output=True, text=True, check=False)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
return tuple(record.read_text().splitlines()) if record.exists() else ()
@ -131,29 +114,137 @@ def _resolve_tf_local(terraform_file: Path, name: str) -> str:
def _entrypoint_argv(dockerfile: Path) -> tuple[str, ...]:
matches = _ENTRYPOINT_RE.findall(dockerfile.read_text())
assert matches, f"no exec-form ENTRYPOINT found in {dockerfile}"
parsed = json.loads(matches[-1])
return tuple(str(part) for part in parsed)
def _cmd_argv(dockerfile: Path) -> tuple[str, ...]:
matches = _CMD_RE.findall(dockerfile.read_text())
assert matches, f"no exec-form CMD found in {dockerfile}"
return tuple(str(part) for part in json.loads(matches[-1]))
def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(
tmp_path: Path,
@pytest.mark.parametrize(
"argv, expected_exec, expected_args",
[
(("proxy", "--config", "/app/config.yaml"), "exec=litellm", "args=--config /app/config.yaml"),
(("gateway",), "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"),
(
("gateway", "--port", "8080"),
"exec=python",
"args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 8080",
),
(("backend",), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"),
(("backend", "--port", "9001"), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001 --port 9001"),
(("ui",), "exec=nginx", "args=-g daemon off;"),
(("migrations",), "exec=python", "args=/app/migrations/run.py"),
(("metrics", "--port", "9090"), "exec=python", "args=-m litellm.proxy.prometheus_metrics_server --port 9090"),
(("collector",), "exec=python", "args=-m litellm.proxy.collector"),
],
)
def test_the_first_argument_selects_the_component(
argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path
) -> None:
recorded = _run_entrypoint(argv, tmp_path)
assert recorded[:2] == (expected_exec, expected_args)
@pytest.mark.parametrize(
"argv, expected",
[
((), ("exec=litellm", "args=")),
(("--port", "4000"), ("exec=litellm", "args=--port 4000")),
(
("--config", "/app/config.yaml", "--detailed_debug"),
("exec=litellm", "args=--config /app/config.yaml --detailed_debug"),
),
],
)
def test_flags_alone_still_run_the_monolithic_proxy(
argv: tuple[str, ...], expected: tuple[str, str], tmp_path: Path
) -> None:
"""Every `docker run litellm --config ...` written before components existed keeps working."""
assert _run_entrypoint(argv, tmp_path)[:2] == expected
def test_the_dockerfile_leaves_the_command_empty_so_the_env_var_can_pick_the_component(tmp_path: Path) -> None:
"""A CMD naming a component would beat LITELLM_COMPONENT, and the proxy already listens on 4000 with no flags."""
dockerfile_text = DOCKERFILE.read_text()
assert _entrypoint_argv(DOCKERFILE) == (IMAGE_ENTRYPOINT_PATH,), "the image ENTRYPOINT must be the bare dispatcher"
assert not _CMD_RE.search(dockerfile_text), "the Dockerfile must not set a CMD"
assert _run_entrypoint((), tmp_path)[:2] == ("exec=litellm", "args=")
@pytest.mark.parametrize(
"component, argv, expected_exec, expected_args",
[
("backend", (), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"),
(
"gateway",
("--port", "4100"),
"exec=python",
"args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 4100",
),
("ui", (), "exec=nginx", "args=-g daemon off;"),
],
)
def test_litellm_component_env_selects_the_component_when_the_command_has_none(
component: str, argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path
) -> None:
"""Helm and Compose can pick the component with an env var and keep `args` for the process flags."""
recorded = _run_entrypoint(argv, tmp_path, LITELLM_COMPONENT=component)
assert recorded[:2] == (expected_exec, expected_args)
def test_an_explicit_component_argument_beats_the_env_var(tmp_path: Path) -> None:
recorded = _run_entrypoint(("backend",), tmp_path, LITELLM_COMPONENT="gateway")
assert recorded[0] == "exec=uvicorn"
def test_the_gateway_honours_num_workers(tmp_path: Path) -> None:
recorded = _run_entrypoint(("gateway",), tmp_path, NUM_WORKERS="4")
assert recorded[1] == "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"
def test_an_unknown_first_word_is_run_as_the_command(tmp_path: Path) -> None:
"""`docker run <image> sh -c ...` and `kubectl exec`-style overrides bypass the dispatcher."""
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
_write_stubs(bin_dir, ("some-tool",))
record = tmp_path / "record.txt"
env = _entrypoint_env(bin_dir, record, {})
result = subprocess.run(
["sh", str(DOCKER_ENTRYPOINT), "some-tool", "--flag"], env=env, capture_output=True, text=True, check=False
)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
assert record.read_text().splitlines()[:2] == ["exec=some-tool", "args=--flag"]
def test_an_unknown_litellm_component_fails_fast_instead_of_guessing(tmp_path: Path) -> None:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
_write_stubs(bin_dir, STUBBED_EXECUTABLES)
record = tmp_path / "record.txt"
env = _entrypoint_env(bin_dir, record, {"LITELLM_COMPONENT": "gatway"})
result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT)], env=env, capture_output=True, text=True, check=False)
assert result.returncode == 64
assert "gatway" in result.stderr
assert not record.exists(), "nothing may be exec'd when the component is unknown"
def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(tmp_path: Path) -> None:
"""`USE_DDTRACE=true` must prefix the command with `ddtrace-run`, turn the openai
integration off, and leave PYTHONPATH alone.
`ddtrace-run` installs its instrumentation by PREPENDING a bootstrap directory to
PYTHONPATH, and the images set PYTHONPATH=/app so the app package is importable. A
wrapper that assigned PYTHONPATH instead of inheriting it would either drop the
bootstrap (silently disabling tracing) or drop /app (breaking the import), so the
recorded value is asserted verbatim.
PYTHONPATH, and the image sets PYTHONPATH=/app so the component packages are
importable. A wrapper that assigned PYTHONPATH instead of inheriting it would either
drop the bootstrap (silently disabling tracing) or drop /app (breaking the import), so
the recorded value is asserted verbatim.
PYTHONPATH_SENTINEL deliberately differs from the images' own /app: with /app as the
PYTHONPATH_SENTINEL deliberately differs from the image's own /app: with /app as the
fixture value, a wrapper that overwrote PYTHONPATH with /app would still satisfy this
assertion and the check would prove nothing.
@ -161,32 +252,22 @@ def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(
it before any litellm code runs, so litellm's in-process `patch_all(..., openai=False)`
can no longer suppress it; leaving it on double-reports every LLM call.
"""
recorded = _run_entrypoint(
COMPONENT_ENTRYPOINT,
("uvicorn", "gateway.main:app", "--workers", "2", "--port", "4000"),
use_ddtrace="true",
tmp_path=tmp_path,
)
recorded = _run_entrypoint(("gateway",), tmp_path, use_ddtrace="true", NUM_WORKERS="2")
assert recorded == (
"exec=ddtrace-run",
"args=uvicorn gateway.main:app --workers 2 --port 4000",
"args=python -m gateway.launch --workers 2 --host 0.0.0.0 --port 4000",
"DD_TRACE_OPENAI_ENABLED=False",
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
)
def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None:
recorded = _run_entrypoint(
COMPONENT_ENTRYPOINT,
("uvicorn", "backend.main:app", "--port", "4001"),
use_ddtrace=None,
tmp_path=tmp_path,
)
recorded = _run_entrypoint(("backend", "--workers", "2"), tmp_path)
assert recorded == (
"exec=uvicorn",
"args=backend.main:app --port 4001",
"args=backend.main:app --host 0.0.0.0 --port 4001 --workers 2",
"DD_TRACE_OPENAI_ENABLED=<unset>",
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
)
@ -196,35 +277,24 @@ def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None:
"use_ddtrace, traced",
[*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)],
)
def test_gating_matches_the_monolithic_entrypoint_and_get_secret_bool(
def test_ddtrace_gating_matches_get_secret_bool_for_every_component(
use_ddtrace: str | None, traced: bool, tmp_path: Path
) -> None:
"""Both entrypoints must accept exactly the spellings `get_secret_bool` accepts.
"""The shell gate must accept exactly the spellings `get_secret_bool` accepts.
`ProxyStartupEvent._init_dd_tracer` reads `USE_DDTRACE` through `get_secret_bool`, which
matches `true` case-insensitively. If the shell gate were stricter, `USE_DDTRACE=True` would
give in-process LLM spans without `ddtrace-run` HTTP spans, a half-enabled state.
"""
component = _run_entrypoint(
COMPONENT_ENTRYPOINT,
("uvicorn", "gateway.main:app"),
use_ddtrace=use_ddtrace,
tmp_path=tmp_path / "component",
)
monolith = _run_entrypoint(
PROD_ENTRYPOINT,
("--port", "4000"),
use_ddtrace=use_ddtrace,
tmp_path=tmp_path / "monolith",
)
component = _run_entrypoint(("gateway",), tmp_path / "component", use_ddtrace=use_ddtrace)
monolith = _run_entrypoint(("--port", "4000"), tmp_path / "monolith", use_ddtrace=use_ddtrace)
expected_exec = "exec=ddtrace-run" if traced else "exec=uvicorn"
expected_openai = "DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=<unset>"
assert component[0] == expected_exec
assert component[0] == ("exec=ddtrace-run" if traced else "exec=python")
assert component[2] == expected_openai
assert monolith[0] == ("exec=ddtrace-run" if traced else "exec=litellm")
assert monolith[2] == expected_openai
assert monolith[1] == ("args=litellm --port 4000" if traced else "args=--port 4000")
assert monolith[2] == expected_openai
def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path) -> None:
@ -236,19 +306,26 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path)
(multiproc_dir / "counter_7.db").write_bytes(b"stale")
(multiproc_dir / "keep.txt").write_text("not a sample")
recorded = _run_entrypoint(("gateway",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir))
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"]
assert recorded[0] == "exec=python"
def test_a_raw_command_leaves_the_prometheus_multiproc_dir_untouched(tmp_path: Path) -> None:
"""The entrypoint cannot tell a raw writer from a raw reader, so `docker run <image> python -m
litellm.proxy.prometheus_metrics_server` must not delete the samples the gateway workers are still writing."""
multiproc_dir = tmp_path / "multiproc"
multiproc_dir.mkdir()
(multiproc_dir / "counter_7.db").write_bytes(b"live")
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
_write_stubs(bin_dir, ("uvicorn",))
_write_stubs(bin_dir, ("python",))
record = tmp_path / "record.txt"
env = {
**os.environ,
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
"RECORD": str(record),
"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir),
}
env.pop("USE_DDTRACE", None)
env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)})
result = subprocess.run(
["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"],
["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "litellm.proxy.prometheus_metrics_server"],
env=env,
capture_output=True,
text=True,
@ -256,161 +333,69 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path)
)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"]
assert record.read_text().splitlines()[0] == "exec=uvicorn"
assert [p.name for p in multiproc_dir.iterdir()] == ["counter_7.db"]
assert record.read_text().splitlines()[0] == "exec=python"
def test_a_raw_command_still_gets_the_prometheus_multiproc_dir_created(tmp_path: Path) -> None:
"""`docker run <image> python -m gateway.launch` cannot write samples into a directory that is not there."""
multiproc_dir = tmp_path / "not-yet" / "multiproc"
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
_write_stubs(bin_dir, ("python",))
record = tmp_path / "record.txt"
env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)})
result = subprocess.run(
["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "gateway.launch"],
env=env,
capture_output=True,
text=True,
check=False,
)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
assert multiproc_dir.is_dir()
assert record.read_text().splitlines()[0] == "exec=python"
@pytest.mark.parametrize("component", ["metrics", "collector"])
def test_readers_of_the_prometheus_multiproc_dir_keep_the_workers_samples(component: str, tmp_path: Path) -> None:
"""The metrics and collector sidecars share the volume with gateway workers that are already
serving traffic, so their (re)start must not erase the samples those workers have written."""
multiproc_dir = tmp_path / "multiproc"
multiproc_dir.mkdir()
(multiproc_dir / "counter_12.db").write_bytes(b"live")
(multiproc_dir / "histogram_13.db").write_bytes(b"live")
recorded = _run_entrypoint((component,), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir))
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["counter_12.db", "histogram_13.db"]
assert recorded[0] == "exec=python"
def test_creates_a_missing_prometheus_multiproc_dir(tmp_path: Path) -> None:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
_write_stubs(bin_dir, ("uvicorn",))
missing = tmp_path / "multiproc"
env = {
**os.environ,
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
"RECORD": str(tmp_path / "record.txt"),
"PROMETHEUS_MULTIPROC_DIR": str(missing),
}
env.pop("USE_DDTRACE", None)
result = subprocess.run(
["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"], env=env, capture_output=True, text=True
)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
_run_entrypoint(("backend",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(missing))
assert missing.is_dir()
def _copied_script(dockerfile: Path, image_path: str) -> Path:
"""Resolve the repo file a Dockerfile `COPY`s to `image_path`, so tests run what the image ships."""
matches = _COPY_RE.findall(dockerfile.read_text())
sources = tuple(src for src, dst in matches if dst == image_path)
assert sources, f"{dockerfile} never COPYs anything to {image_path}"
source = REPO_ROOT / sources[-1]
assert source.is_file(), f"{dockerfile} COPYs {sources[-1]}, which does not exist in the build context"
return source
@pytest.mark.parametrize(
"use_ddtrace, traced",
[*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)],
)
def test_build_from_pip_image_launches_litellm_through_the_prod_entrypoint(
use_ddtrace: str | None, traced: bool, tmp_path: Path
) -> None:
"""Run the build_from_pip image's ENTRYPOINT + CMD through the script it actually COPYs.
The image used to `ENTRYPOINT ["litellm"]`, so `USE_DDTRACE` was inert there at any spelling.
Resolving the ENTRYPOINT path back to its COPY source and executing it with the Dockerfile's
CMD checks the launch the container performs, not just that the Dockerfile mentions the script.
"""
entrypoint = _entrypoint_argv(BUILD_FROM_PIP_DOCKERFILE)
assert len(entrypoint) == 1, (
f"{BUILD_FROM_PIP_DOCKERFILE} ENTRYPOINT must be the bare script so CMD reaches litellm"
)
script = _copied_script(BUILD_FROM_PIP_DOCKERFILE, entrypoint[0])
assert script == PROD_ENTRYPOINT, f"{BUILD_FROM_PIP_DOCKERFILE} bypasses the ddtrace-aware entrypoint"
assert f"chmod +x {entrypoint[0]}" in BUILD_FROM_PIP_DOCKERFILE.read_text()
cmd = _cmd_argv(BUILD_FROM_PIP_DOCKERFILE)
recorded = _run_entrypoint(script, cmd, use_ddtrace=use_ddtrace, tmp_path=tmp_path)
cmd_str = " ".join(cmd)
assert recorded == (
"exec=ddtrace-run" if traced else "exec=litellm",
f"args=litellm {cmd_str}" if traced else f"args={cmd_str}",
"DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=<unset>",
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
)
def test_entrypoint_script_is_executable() -> None:
mode = COMPONENT_ENTRYPOINT.stat().st_mode
assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT"
assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user"
def test_entrypoint_script_has_no_carriage_returns() -> None:
assert b"\r" not in COMPONENT_ENTRYPOINT.read_bytes()
@pytest.mark.parametrize(
"dockerfile, launcher",
[
(GATEWAY_DOCKERFILE, "python -m gateway.launch"),
(BACKEND_DOCKERFILE, "uvicorn backend.main:app"),
],
)
def test_component_images_launch_uvicorn_through_the_entrypoint(dockerfile: Path, launcher: str) -> None:
entrypoint = " ".join(_entrypoint_argv(dockerfile))
assert IMAGE_ENTRYPOINT_PATH in entrypoint, f"{dockerfile} bypasses the ddtrace-aware entrypoint"
assert launcher in entrypoint
assert entrypoint.index(IMAGE_ENTRYPOINT_PATH) < entrypoint.index(launcher), (
f"{dockerfile} must invoke uvicorn through the entrypoint, not the other way around"
)
@pytest.mark.parametrize(
"use_ddtrace, num_workers, expected_exec, expected_args",
[
(None, "4", "exec=python", "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"),
(None, None, "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"),
("true", "4", "exec=ddtrace-run", "args=python -m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"),
],
)
def test_gateway_image_execs_the_supervisor_with_its_worker_count(
use_ddtrace: str | None, num_workers: str | None, expected_exec: str, expected_args: str, tmp_path: Path
) -> None:
"""Run the gateway image's ENTRYPOINT + CMD and record what the container execs.
The Dockerfile's `/app/...` script path is resolved to the checked-in script and `python`
is stubbed on PATH, so the assertion is on the argv `gateway.launch` receives, not on the
Dockerfile text.
"""
entrypoint = tuple(
part.replace(IMAGE_ENTRYPOINT_PATH, str(COMPONENT_ENTRYPOINT)) for part in _entrypoint_argv(GATEWAY_DOCKERFILE)
)
bin_dir = tmp_path / "bin"
bin_dir.mkdir(parents=True)
_write_stubs(bin_dir, ("ddtrace-run", "python", "uvicorn"))
record = tmp_path / "record.txt"
overrides = {"USE_DDTRACE": use_ddtrace, "NUM_WORKERS": num_workers}
env = {
**{k: v for k, v in os.environ.items() if k not in ("DD_TRACE_OPENAI_ENABLED", *overrides)},
**{k: v for k, v in overrides.items() if v is not None},
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
"RECORD": str(record),
"PYTHONPATH": PYTHONPATH_SENTINEL,
}
result = subprocess.run(
[*entrypoint, *_cmd_argv(GATEWAY_DOCKERFILE)], env=env, capture_output=True, text=True, check=False
)
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
assert tuple(record.read_text().splitlines())[:2] == (expected_exec, expected_args)
@pytest.mark.parametrize("dockerfile", [GATEWAY_DOCKERFILE, BACKEND_DOCKERFILE])
def test_component_images_make_the_entrypoint_executable(dockerfile: Path) -> None:
body = dockerfile.read_text()
assert "chmod +x docker/component_entrypoint.sh" in body
@pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name)
@pytest.mark.parametrize("component", ["gateway", "backend"])
@pytest.mark.parametrize("use_ddtrace", [*TRUTHY_USE_DDTRACE, *FALSY_USE_DDTRACE])
def test_terraform_launch_command_matches_the_script_contract(
terraform_file: Path, component: str, use_ddtrace: str | None, tmp_path: Path
) -> None:
"""The Terraform command and `docker/component_entrypoint.sh` must decide identically.
"""The Terraform command and `docker-entrypoint.sh <component>` must decide identically.
The decision deliberately lives in two places. The script is what the image ENTRYPOINT runs;
the Terraform strings are what runs when a deployment overrides that ENTRYPOINT, and they
cannot call the script because the caller supplies the image tag and it may predate the file.
Both modules default to a tag that does. So instead of asserting a shared path, this runs both
implementations under the same environment and asserts they agree on which binary is exec'd
and on whether the openai integration is disabled.
So instead of asserting a shared path, this runs both implementations under the same
environment and asserts they agree on which binary is exec'd and on whether the openai
integration is disabled.
"""
launcher = COMPONENT_LAUNCHERS[component]
app_target = " ".join(launcher[1:])
@ -421,18 +406,14 @@ def test_terraform_launch_command_matches_the_script_contract(
_write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python"))
from_terraform = _run_shell_command(command, bin_dir, tmp_path / "terraform.txt", use_ddtrace)
from_script = _run_entrypoint(
COMPONENT_ENTRYPOINT,
launcher,
use_ddtrace=use_ddtrace,
tmp_path=tmp_path / "script",
)
from_script = _run_entrypoint((component,), tmp_path / "script", use_ddtrace=use_ddtrace)
assert from_terraform[0] == from_script[0], (
f"{terraform_file} disagrees with the script on USE_DDTRACE={use_ddtrace}"
)
assert from_terraform[2] == from_script[2], f"{terraform_file} disagrees with the script on the openai integration"
assert app_target in from_terraform[1]
assert app_target in from_script[1]
assert "gateway.main:app" not in from_terraform[1], f"{terraform_file} bypasses the gateway.launch supervisor"
if use_ddtrace in TRUTHY_USE_DDTRACE:
@ -478,9 +459,11 @@ def test_terraform_does_not_depend_on_the_entrypoint_script(terraform_file: Path
)
def test_gateway_keeps_its_worker_count_and_backend_keeps_a_single_process() -> None:
gateway = " ".join(_entrypoint_argv(GATEWAY_DOCKERFILE)) + " " + " ".join(_cmd_argv(GATEWAY_DOCKERFILE))
backend = " ".join(_entrypoint_argv(BACKEND_DOCKERFILE)) + " " + " ".join(_cmd_argv(BACKEND_DOCKERFILE))
def test_entrypoint_script_is_executable() -> None:
mode = DOCKER_ENTRYPOINT.stat().st_mode
assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT"
assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user"
assert "--workers" in gateway and "NUM_WORKERS" in gateway
assert "--workers" not in backend and "NUM_WORKERS" not in backend
def test_entrypoint_script_has_no_carriage_returns() -> None:
assert b"\r" not in DOCKER_ENTRYPOINT.read_bytes()

View file

@ -1,5 +1,5 @@
"""
Static checks that every proxy Docker image installs the `bedrock-realtime` extra.
Static checks that the shipped Docker image installs the `bedrock-realtime` extra.
Bedrock Nova Sonic speech-to-speech (`/v1/realtime`) needs `aws-sdk-bedrock-runtime`,
which only ships in the `bedrock-realtime` extra. An image whose `uv sync` stages
@ -23,12 +23,7 @@ else:
REPO_ROOT: Final = os.path.join(os.path.dirname(__file__), "..", "..")
PROXY_DOCKERFILES: Final = (
"Dockerfile",
os.path.join("docker", "Dockerfile.non_root"),
os.path.join("docker", "Dockerfile.database"),
os.path.join("gateway", "Dockerfile"),
)
PROXY_DOCKERFILES: Final = ("Dockerfile",)
CONTINUED_LINE_RE: Final = re.compile(r"(?:\\\n|[^\n])+")
UV_SYNC_BOUNDARY_RE: Final = re.compile(r"(?=uv sync)")

View file

@ -1,39 +1,26 @@
"""
Static checks on docker/Dockerfile.non_root.
Static checks on the shipped Dockerfile's runtime user.
The non_root image is intended for deployment into hardened Kubernetes
clusters where `securityContext.runAsNonRoot: true` is enforced. The
kubelet validates non-root status by parsing the image's USER field as
an integer — a string name like "nobody" is rejected with
CreateContainerConfigError because the kubelet cannot resolve
/etc/passwd inside the image at admission time.
The image is deployed into hardened Kubernetes clusters where
`securityContext.runAsNonRoot: true` is enforced. The kubelet validates
non-root status by parsing the image's USER field as an integer: a string
name like "nonroot" is rejected with CreateContainerConfigError because the
kubelet cannot resolve /etc/passwd inside the image at admission time.
"""
import os
import re
import pytest
DOCKERFILE_PATH = os.path.join(
os.path.dirname(__file__),
"..",
"..",
"docker",
"Dockerfile.non_root",
)
DOCKERFILE_PATH = os.path.join(os.path.dirname(__file__), "..", "..", "Dockerfile")
def _final_user_directive(dockerfile_text: str) -> str:
"""Return the value of the last `USER` directive in the file."""
"""Return the uid of the last `USER` directive in the file (`USER uid` or `USER uid:gid`)."""
matches = re.findall(r"^USER\s+(\S+)\s*$", dockerfile_text, re.MULTILINE)
assert matches, "Dockerfile.non_root has no USER directive"
return matches[-1]
assert matches, "Dockerfile has no USER directive"
return matches[-1].split(":", 1)[0]
@pytest.mark.skipif(
not os.path.exists(DOCKERFILE_PATH),
reason="Dockerfile.non_root not present in this checkout",
)
def test_final_user_directive_is_numeric():
"""The runtime USER must be a numeric UID so kubelet's runAsNonRoot
admission check (strconv.Atoi) succeeds."""
@ -43,12 +30,9 @@ def test_final_user_directive_is_numeric():
final_user = _final_user_directive(contents)
assert final_user.isdigit(), (
f"Dockerfile.non_root final USER is {final_user!r}; must be a numeric UID "
f"Dockerfile final USER is {final_user!r}; must be a numeric UID "
"so Kubernetes' runAsNonRoot admission check can verify non-root status. "
"See https://kubernetes.io/docs/tasks/configure-pod-container/security-context/"
)
assert int(final_user) != 0, (
f"Dockerfile.non_root final USER is {final_user} (root); the non_root image "
"must run as a non-zero UID."
)
assert int(final_user) != 0, f"Dockerfile final USER is {final_user} (root); the image must run as a non-zero UID."

View file

@ -1,42 +0,0 @@
# syntax=docker/dockerfile:1.7
# UI container — Next.js static export served by nginx.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
ARG NGINX_VERSION=1.31.5-alpine3.24@sha256:34f40471dea485273c5e2a04dd5e97a682332ceb4a9adecd67de450dcb2fb390
# ---------- builder ----------
FROM ${UI_BUILD_IMAGE} AS builder
ENV NEXT_TELEMETRY_DISABLED=1 \
npm_config_fund=false \
npm_config_audit=false
WORKDIR /app
# Layer the lockfile-only install above the source copy so source-only
# edits don't bust the install cache.
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
RUN --mount=type=cache,target=/root/.npm \
npm ci --prefer-offline
COPY ui/litellm-dashboard/ ./
RUN npm run build
# ---------- runtime ----------
FROM nginx:${NGINX_VERSION} AS runtime
# Drop the upstream default :80 server; we own the config.
RUN rm -f /etc/nginx/conf.d/default.conf
# Static export → web root.
COPY --from=builder /app/out /usr/share/nginx/html
# Routing rules — see ui/nginx.conf for the full description.
COPY ui/nginx.conf /etc/nginx/nginx.conf
EXPOSE 3000/tcp
# nginx as PID 1 in foreground; respects SIGTERM out of the box, so
# no tini/dumb-init wrapper needed.
CMD ["nginx", "-g", "daemon off;"]

View file

@ -5,6 +5,7 @@ worker_processes auto;
# nginx image's /var/cache/nginx and /run are root-owned 755) and works
# with readOnlyRootFilesystem when /tmp is an emptyDir.
pid /tmp/nginx.pid;
error_log /dev/stderr warn;
events { worker_connections 1024; }
@ -15,6 +16,7 @@ http {
uwsgi_temp_path /tmp/nginx-uwsgi-temp;
scgi_temp_path /tmp/nginx-scgi-temp;
access_log /dev/stdout;
include /etc/nginx/mime.types;
default_type application/octet-stream;
sendfile on;
@ -29,7 +31,6 @@ http {
application/javascript
application/json
text/css
text/html
image/svg+xml
font/woff
font/woff2;
@ -37,16 +38,16 @@ http {
server {
listen 3000 default_server;
server_name _;
root /usr/share/nginx/html;
root /var/lib/litellm/ui;
# next.config.mjs sets assetPrefix=/litellm-asset-prefix, which makes
# the built HTML reference /litellm-asset-prefix/_next/... — but the
# static export only emits files under /_next/. Map the prefix to
# the real tree at request time instead of duplicating the directory
# at build time. NB: alias rewrites the location prefix, so
# /litellm-asset-prefix/_next/foo.js → /usr/share/nginx/html/_next/foo.js.
# /litellm-asset-prefix/_next/foo.js → /var/lib/litellm/ui/_next/foo.js.
location /litellm-asset-prefix/_next/ {
alias /usr/share/nginx/html/_next/;
alias /var/lib/litellm/ui/_next/;
expires 1y;
add_header Cache-Control "public, immutable";
}
@ -63,7 +64,7 @@ http {
add_header Cache-Control "public, immutable";
}
location ^~ /ui/assets/ {
alias /usr/share/nginx/html/assets/;
alias /var/lib/litellm/ui/assets/;
}
location = /favicon.ico {
try_files $uri =404;