mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat(docker): one shipped image, one entrypoint beside the Dockerfile, one compose file
Some checks failed
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
Some checks failed
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
Delete the nine other Dockerfiles and ship the root Dockerfile as the one non-root image. /app/docker-entrypoint.sh, beside the Dockerfile, is the only entrypoint script and dispatches proxy, gateway, backend, ui, migrations, metrics and collector by first argument or LITELLM_COMPONENT. The root docker-compose.yml absorbs the quickstart and hardened files: it pulls the published image on up and builds on up --build, requires LITELLM_MASTER_KEY and LITELLM_SALT_KEY from .env, runs read-only with all capabilities dropped, and keeps Prometheus behind the monitoring profile. CI, docs and tests point at the one image; dead helper scripts are removed. The metrics and collector sidecars only read PROMETHEUS_MULTIPROC_DIR, so the entrypoint creates it for them and for raw commands without deleting the workers' samples Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
a6f6c64b6e
commit
14bcbd03cf
43 changed files with 738 additions and 1765 deletions
|
|
@ -532,11 +532,10 @@ jobs:
|
|||
key: v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- save_cargo_target
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
name: Run prisma migrations
|
||||
command: |
|
||||
set +e
|
||||
chmod +x docker/entrypoint.sh
|
||||
./docker/entrypoint.sh
|
||||
uv run --no-sync python litellm/proxy/prisma_migration.py
|
||||
set -e
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
|
|
@ -606,11 +605,10 @@ jobs:
|
|||
- ~/.cache/uv
|
||||
key: v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
name: Run prisma migrations
|
||||
command: |
|
||||
set +e
|
||||
chmod +x docker/entrypoint.sh
|
||||
./docker/entrypoint.sh
|
||||
uv run --no-sync python litellm/proxy/prisma_migration.py
|
||||
set -e
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
|
|
@ -681,11 +679,10 @@ jobs:
|
|||
- ~/.cache/uv
|
||||
key: v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
name: Run prisma migrations
|
||||
command: |
|
||||
set +e
|
||||
chmod +x docker/entrypoint.sh
|
||||
./docker/entrypoint.sh
|
||||
uv run --no-sync python litellm/proxy/prisma_migration.py
|
||||
set -e
|
||||
|
||||
# Run pytest and generate JUnit XML report
|
||||
|
|
@ -2422,16 +2419,18 @@ jobs:
|
|||
path: test-results
|
||||
|
||||
proxy_build_from_pip_tests:
|
||||
# Change from docker to machine executor
|
||||
# Validates the published PyPI artifact, not the checked-out source tree: the
|
||||
# proxy is pip-installed from PyPI into its own venv and booted on the runner
|
||||
machine:
|
||||
image: ubuntu-2204:2024.04.1
|
||||
resource_class: large
|
||||
working_directory: ~/project
|
||||
environment:
|
||||
LITELLM_PIP_VERSION: "1.83.0"
|
||||
steps:
|
||||
- checkout
|
||||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
# Remove Docker CLI installation since it's already available in machine executor
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
|
|
@ -2439,47 +2438,53 @@ jobs:
|
|||
command: |
|
||||
uv sync --frozen --all-groups --all-extras --python 3.12
|
||||
- run:
|
||||
name: Build Docker image
|
||||
name: Install the published litellm proxy from PyPI into its own venv
|
||||
command: |
|
||||
docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip .
|
||||
uv venv --python 3.13 /tmp/litellm-pip
|
||||
uv pip install --python /tmp/litellm-pip/bin/python \
|
||||
"litellm[proxy,proxy-runtime]==${LITELLM_PIP_VERSION}" \
|
||||
"google-cloud-aiplatform==1.133.0" \
|
||||
"google-genai==1.37.0" \
|
||||
"anthropic[vertex]==0.84.0" \
|
||||
"grpcio==1.78.0" \
|
||||
"prometheus-client==0.20.0" \
|
||||
"langfuse==2.59.7" \
|
||||
"opentelemetry-api==1.28.0" \
|
||||
"opentelemetry-sdk==1.28.0" \
|
||||
"opentelemetry-exporter-otlp==1.28.0" \
|
||||
"ddtrace==4.11.0" \
|
||||
"sentry-sdk==2.21.0" \
|
||||
"mangum==0.17.0" \
|
||||
"azure-ai-contentsafety==1.0.0" \
|
||||
"azure-storage-file-datalake==12.20.0" \
|
||||
"pypdf==6.7.5" \
|
||||
"llm-sandbox==0.3.31" \
|
||||
"detect-secrets==1.5.0" \
|
||||
"prisma==0.11.0" \
|
||||
"openai==2.24.0"
|
||||
/tmp/litellm-pip/bin/python -c "import litellm, sys; print('litellm', litellm.__version__ if hasattr(litellm, '__version__') else '', sys.version)"
|
||||
- start_postgres
|
||||
- start_fake_openai_endpoint
|
||||
- run:
|
||||
name: Run Docker container
|
||||
name: Run the published proxy
|
||||
# intentionally give bad redis credentials here
|
||||
# the OTEL test - should get this as a trace
|
||||
command: |
|
||||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true \
|
||||
-e DATABASE_URL=postgresql://postgres:postgres@host.docker.internal:5432/circle_test \
|
||||
-e REDIS_HOST=$REDIS_HOST \
|
||||
-e REDIS_PASSWORD=$REDIS_PASSWORD \
|
||||
-e REDIS_PORT=$REDIS_PORT \
|
||||
-e LITELLM_MASTER_KEY="sk-1234" \
|
||||
-e OPENAI_API_KEY=$OPENAI_API_KEY \
|
||||
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
|
||||
-e LITELLM_LICENSE=$LITELLM_LICENSE \
|
||||
-e OTEL_EXPORTER="in_memory" \
|
||||
-e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \
|
||||
-e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
|
||||
-e AWS_REGION_NAME=$AWS_REGION_NAME \
|
||||
-e COHERE_API_KEY=$COHERE_API_KEY \
|
||||
-e USE_DDTRACE=True \
|
||||
-e DD_API_KEY=$DD_API_KEY \
|
||||
-e DD_SITE=$DD_SITE \
|
||||
-e GCS_FLUSH_INTERVAL="1" \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \
|
||||
my-app:latest \
|
||||
--config /app/config.yaml \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
background: true
|
||||
command: |
|
||||
export LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true
|
||||
export DATABASE_URL=postgresql://postgres:postgres@localhost:5432/circle_test
|
||||
export LITELLM_MASTER_KEY="sk-1234"
|
||||
export LITELLM_LICENSE="bad-license"
|
||||
export FAKE_OPENAI_API_BASE=http://localhost:8190
|
||||
export OTEL_EXPORTER="in_memory"
|
||||
export USE_DDTRACE=True
|
||||
export DD_TRACE_OPENAI_ENABLED="False"
|
||||
export GCS_FLUSH_INTERVAL="1"
|
||||
export LITELLM_LOG=ERROR
|
||||
cd /tmp/litellm-pip
|
||||
exec /tmp/litellm-pip/bin/ddtrace-run /tmp/litellm-pip/bin/litellm \
|
||||
--config ~/project/tests/basic_proxy_startup_tests/build_from_pip_config.yaml \
|
||||
--port 4000 2>&1 | tee /tmp/litellm-pip.log
|
||||
- wait_for_service:
|
||||
url: http://localhost:4000
|
||||
timeout: "300"
|
||||
|
|
@ -2495,14 +2500,15 @@ jobs:
|
|||
--junitxml=test-results/junit-2.xml \
|
||||
--durations=5"
|
||||
no_output_timeout: 15m
|
||||
# Clean up first container
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- run:
|
||||
name: Stop and remove first container
|
||||
name: Proxy log
|
||||
command: cat /tmp/litellm-pip.log || true
|
||||
when: always
|
||||
- run:
|
||||
name: Stop postgres
|
||||
command: |
|
||||
docker stop my-app || true
|
||||
docker rm my-app || true
|
||||
docker stop postgres-db || true
|
||||
docker rm postgres-db || true
|
||||
when: always
|
||||
|
|
@ -3023,7 +3029,7 @@ jobs:
|
|||
docker build \
|
||||
--label org.opencontainers.image.revision="$(git rev-parse HEAD)" \
|
||||
-t litellm-docker-database:ci \
|
||||
-f docker/Dockerfile.database .
|
||||
-f Dockerfile .
|
||||
fi
|
||||
python3 .circleci/scripts/run_migration_tests.py record-image
|
||||
|
||||
|
|
|
|||
|
|
@ -7,7 +7,6 @@ tests
|
|||
.devcontainer
|
||||
*.tgz
|
||||
log.txt
|
||||
docker/Dockerfile.*
|
||||
|
||||
# Claude Flow generated files (must be excluded from Docker build)
|
||||
.claude/
|
||||
|
|
|
|||
9
.github/ci-coverage-allowlist.yml
vendored
9
.github/ci-coverage-allowlist.yml
vendored
|
|
@ -102,15 +102,6 @@ test_paths:
|
|||
- tests/integration/test_oci_proxy_integration.py
|
||||
|
||||
dockerfiles:
|
||||
- reason: >-
|
||||
The dashboard container is a static Next.js export served by nginx, and the dashboard build
|
||||
and lint workflows already exercise that output, so building the image adds no signal about it
|
||||
paths:
|
||||
- ui/Dockerfile
|
||||
- reason: >-
|
||||
An example image under cookbook/ that is documentation rather than a shipped artifact
|
||||
paths:
|
||||
- cookbook/litellm-ollama-docker-image/Dockerfile
|
||||
- reason: >-
|
||||
The Rust gateway image compiles the whole workspace in release mode, which is too slow for
|
||||
a per-pull-request job while the gateway binary is still being assembled; the Rust lint,
|
||||
|
|
|
|||
33
.github/workflows/compat-matrix-image.yml
vendored
33
.github/workflows/compat-matrix-image.yml
vendored
|
|
@ -26,17 +26,24 @@ jobs:
|
|||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build the Render cron image
|
||||
run: docker build -f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix:${{ github.sha }} tests/e2e
|
||||
|
||||
- name: Resolve and install the Claude Code CLI as the cron user
|
||||
- name: Install the pinned uv the cron image ships
|
||||
env:
|
||||
UV_VERSION: 0.10.9
|
||||
UV_SHA256: 20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
|
||||
run: |
|
||||
docker run --rm compat-matrix:${{ github.sha }} bash -c '
|
||||
set -euo pipefail
|
||||
whoami
|
||||
gh --version
|
||||
uv --version
|
||||
version="$(uv run --no-project --python 3.12 python /opt/litellm/tests/e2e/claude_code/pr_gate_version_resolver.py)"
|
||||
/opt/litellm/tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" /tmp/claude-cli
|
||||
/tmp/claude-cli/claude --version
|
||||
'
|
||||
set -euo pipefail
|
||||
curl -fsSLo "$RUNNER_TEMP/uv.tar.gz" "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz"
|
||||
echo "${UV_SHA256} $RUNNER_TEMP/uv.tar.gz" | sha256sum -c -
|
||||
mkdir -p "$RUNNER_TEMP/bin"
|
||||
tar -xzf "$RUNNER_TEMP/uv.tar.gz" -C "$RUNNER_TEMP/bin" --strip-components=1 uv-x86_64-unknown-linux-gnu/uv
|
||||
echo "$RUNNER_TEMP/bin" >> "$GITHUB_PATH"
|
||||
|
||||
- name: Resolve and install the Claude Code CLI as an unprivileged user
|
||||
run: |
|
||||
set -euo pipefail
|
||||
whoami
|
||||
gh --version
|
||||
uv --version
|
||||
version="$(uv run --no-project --python 3.12 python tests/e2e/claude_code/pr_gate_version_resolver.py)"
|
||||
tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" "$RUNNER_TEMP/claude-cli"
|
||||
"$RUNNER_TEMP/claude-cli/claude" --version
|
||||
|
|
|
|||
277
.github/workflows/image-scan.yml
vendored
277
.github/workflows/image-scan.yml
vendored
|
|
@ -8,21 +8,17 @@ on:
|
|||
- "litellm_**"
|
||||
paths:
|
||||
- Dockerfile
|
||||
- docker/Dockerfile.non_root
|
||||
- migrations/Dockerfile
|
||||
- .dockerignore
|
||||
- docker-entrypoint.sh
|
||||
- migrations/run.py
|
||||
- gateway/Dockerfile
|
||||
- gateway/main.py
|
||||
- backend/Dockerfile
|
||||
- gateway/launch.py
|
||||
- backend/main.py
|
||||
- docker/component_entrypoint.sh
|
||||
- docker/entrypoint.sh
|
||||
- litellm/proxy/prisma_migration.py
|
||||
- litellm-proxy-extras/**
|
||||
- tests/proxy_migration_tests/**
|
||||
- uv.lock
|
||||
- ui/litellm-dashboard/package-lock.json
|
||||
- ui/Dockerfile
|
||||
- ui/nginx.conf
|
||||
- .github/workflows/image-scan.yml
|
||||
- .grype.yaml
|
||||
|
|
@ -36,9 +32,12 @@ concurrency:
|
|||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
IMAGE: litellm-image-scan:${{ github.sha }}
|
||||
|
||||
jobs:
|
||||
image-scan:
|
||||
name: image-scan
|
||||
build:
|
||||
name: build
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
|
|
@ -51,7 +50,92 @@ jobs:
|
|||
with:
|
||||
persist-credentials: false
|
||||
|
||||
# One image ships; every component below is a mode of it. Built once
|
||||
# here and handed to the verify matrix so a component check never runs
|
||||
# against a different build than the scan.
|
||||
- name: Build the image
|
||||
run: docker build -t "$IMAGE" .
|
||||
|
||||
- name: Save the image for the verify matrix
|
||||
run: docker save "$IMAGE" | zstd -T0 -3 -o "$RUNNER_TEMP/litellm-image.tar.zst"
|
||||
|
||||
- uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1
|
||||
with:
|
||||
name: litellm-image
|
||||
path: ${{ runner.temp }}/litellm-image.tar.zst
|
||||
retention-days: 1
|
||||
compression-level: 0
|
||||
|
||||
verify:
|
||||
name: ${{ matrix.check }}
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
check: [scan, migrations, gateway, backend, ui]
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
|
||||
with:
|
||||
name: litellm-image
|
||||
path: ${{ runner.temp }}
|
||||
|
||||
- name: Load the image
|
||||
run: zstd -d --stdout "$RUNNER_TEMP/litellm-image.tar.zst" | docker load
|
||||
|
||||
- name: Set up Python
|
||||
if: matrix.check != 'scan'
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Install pytest
|
||||
if: matrix.check != 'scan'
|
||||
run: python -m pip install "pytest==9.0.3"
|
||||
|
||||
# The prisma bake must migrate a fresh DB with no egress as an arbitrary
|
||||
# non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyFilesystem).
|
||||
# `docker run` as the default uid with network hides a broken bake because
|
||||
# the migration entrypoint exits 0 even when it applied nothing; asserting
|
||||
# the schema was created is what catches it.
|
||||
- name: Verify offline migration as a non-root uid
|
||||
if: matrix.check == 'migrations'
|
||||
env:
|
||||
LITELLM_IMAGE: ${{ env.IMAGE }}
|
||||
run: |
|
||||
python -m pytest -v \
|
||||
tests/proxy_migration_tests/test_offline_image_migration.py \
|
||||
tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py
|
||||
|
||||
- name: Verify the gateway component serves offline as a non-root uid
|
||||
if: matrix.check == 'gateway'
|
||||
env:
|
||||
LITELLM_IMAGE: ${{ env.IMAGE }}
|
||||
LITELLM_COMPONENT: gateway
|
||||
run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py
|
||||
|
||||
- name: Verify the backend component serves offline as a non-root uid
|
||||
if: matrix.check == 'backend'
|
||||
env:
|
||||
LITELLM_IMAGE: ${{ env.IMAGE }}
|
||||
LITELLM_COMPONENT: backend
|
||||
run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py
|
||||
|
||||
- name: Verify the ui component serves as an arbitrary uid with a read-only root fs
|
||||
if: matrix.check == 'ui'
|
||||
env:
|
||||
LITELLM_IMAGE: ${{ env.IMAGE }}
|
||||
run: python -m pytest -v tests/proxy_migration_tests/test_ui_image_serves_offline.py
|
||||
|
||||
- name: Download Grype v0.114.0
|
||||
if: matrix.check == 'scan'
|
||||
run: |
|
||||
curl -fsSL --retry 3 -o "$RUNNER_TEMP/grype.tar.gz" \
|
||||
https://github.com/anchore/grype/releases/download/v0.114.0/grype_0.114.0_linux_amd64.tar.gz
|
||||
|
|
@ -59,29 +143,6 @@ jobs:
|
|||
tar xzf "$RUNNER_TEMP/grype.tar.gz" -C "$RUNNER_TEMP" grype
|
||||
chmod +x "$RUNNER_TEMP/grype"
|
||||
|
||||
# Dockerfile.non_root is the rootless variant we ship. The other
|
||||
# Dockerfiles share the same wolfi base and apk set, so OS-layer coverage
|
||||
# is the same; matrix-scan if those variants ever diverge.
|
||||
- name: Build runtime image
|
||||
run: docker build -f docker/Dockerfile.non_root -t litellm-image-scan:${{ github.sha }} .
|
||||
|
||||
# The prisma bake must migrate a fresh DB with no egress as an arbitrary
|
||||
# non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyRootFilesystem).
|
||||
# `docker run` as the default uid with network hides a broken bake because
|
||||
# the migration entrypoint exits 0 even when it applied nothing; asserting
|
||||
# the schema was created is what catches it.
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify offline migration as a non-root uid
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-image-scan:${{ github.sha }}
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
|
||||
|
||||
# Scans the whole shipped artifact: OS/apk plus every language package
|
||||
# baked into the image, including ones no lockfile declares (e.g. prisma's
|
||||
# vendored node engine) that osv-scan cannot see. osv-scan stays the fast
|
||||
|
|
@ -89,160 +150,12 @@ jobs:
|
|||
# free OSS, run as a pinned, checksum-verified binary; no GitHub Action
|
||||
# dependency and no vendor SaaS callout.
|
||||
- name: Scan image for fixable HIGH/CRITICAL CVEs
|
||||
if: matrix.check == 'scan'
|
||||
env:
|
||||
GRYPE_MATCH_PYTHON_USING_CPES: "true"
|
||||
run: |
|
||||
"$RUNNER_TEMP/grype" litellm-image-scan:${{ github.sha }} \
|
||||
"$RUNNER_TEMP/grype" "$IMAGE" \
|
||||
--config .grype.yaml \
|
||||
--only-fixed \
|
||||
--fail-on high \
|
||||
--output table
|
||||
|
||||
runtime-image:
|
||||
name: runtime-image
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
github.event.pull_request.head.repo.full_name == github.repository
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build runtime image
|
||||
run: docker build -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify offline migration as a non-root uid
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-runtime-scan:${{ github.sha }}
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
|
||||
|
||||
migrations-image:
|
||||
name: migrations-image
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
github.event.pull_request.head.repo.full_name == github.repository
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build migrations image
|
||||
run: docker build -f migrations/Dockerfile -t litellm-migrations-scan:${{ github.sha }} .
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify offline migration as a non-root uid
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-migrations-scan:${{ github.sha }}
|
||||
LITELLM_MIGRATION_INTERPRETER: python3
|
||||
LITELLM_MIGRATION_SCRIPT: /app/run.py
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py -v
|
||||
|
||||
gateway-image:
|
||||
name: gateway-image
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
github.event.pull_request.head.repo.full_name == github.repository
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build gateway image
|
||||
run: docker build -f gateway/Dockerfile -t litellm-gateway-scan:${{ github.sha }} .
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify the gateway serves offline as a non-root uid
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-gateway-scan:${{ github.sha }}
|
||||
LITELLM_COMPONENT_PORT: "4000"
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
|
||||
|
||||
ui-image:
|
||||
name: ui-image
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
github.event.pull_request.head.repo.full_name == github.repository
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build UI image
|
||||
run: docker build -f ui/Dockerfile -t litellm-ui-scan:${{ github.sha }} .
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify the UI serves offline as an arbitrary uid with a read-only root fs
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-ui-scan:${{ github.sha }}
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_ui_image_serves_offline.py -v
|
||||
|
||||
backend-image:
|
||||
name: backend-image
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name != 'pull_request' ||
|
||||
github.event.pull_request.head.repo.full_name == github.repository
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Build backend image
|
||||
run: docker build -f backend/Dockerfile -t litellm-backend-scan:${{ github.sha }} .
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Verify the backend serves offline as a non-root uid
|
||||
env:
|
||||
LITELLM_IMAGE: litellm-backend-scan:${{ github.sha }}
|
||||
LITELLM_COMPONENT_PORT: "4001"
|
||||
run: |
|
||||
python -m pip install "pytest==9.0.3"
|
||||
python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py -v
|
||||
|
|
|
|||
4
.github/workflows/test-litellm-ui-build.yml
vendored
4
.github/workflows/test-litellm-ui-build.yml
vendored
|
|
@ -33,8 +33,8 @@ jobs:
|
|||
# Built through the image stage rather than the checkout, because the
|
||||
# stage copies ui/litellm-dashboard/ alone: an import reaching above the
|
||||
# dashboard root resolves in a checkout and fails in every image we ship.
|
||||
# Dockerfile, docker/Dockerfile.non_root and ui/Dockerfile share this
|
||||
# stage verbatim, so building one covers all three.
|
||||
# The shipped image serves this stage's output in every component, so
|
||||
# building the stage here covers what ships.
|
||||
- name: Build the dashboard as the shipped images build it
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
run: docker build --target ui-builder -f Dockerfile .
|
||||
|
|
|
|||
|
|
@ -265,8 +265,8 @@ uv run litellm --config your_config.yaml
|
|||
If you want to build the Docker image yourself:
|
||||
|
||||
```bash
|
||||
# Build using the non-root Dockerfile
|
||||
docker build -f docker/Dockerfile.non_root -t litellm_dev .
|
||||
# Build the image (runs as a non-root user by default)
|
||||
docker build -t litellm_dev .
|
||||
|
||||
# Generate a master key. Requests send it as the bearer token
|
||||
export LITELLM_MASTER_KEY="sk-$(openssl rand -hex 32)"
|
||||
|
|
|
|||
68
Dockerfile
68
Dockerfile
|
|
@ -111,8 +111,12 @@ RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/
|
|||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
|
||||
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||
RUN sed -i 's/\r$//' docker-entrypoint.sh && chmod +x docker-entrypoint.sh
|
||||
|
||||
RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
cp -r litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \
|
||||
cp litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \
|
||||
touch /var/lib/litellm/ui/.litellm_ui_ready
|
||||
|
||||
# Runtime stage
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
|
@ -125,23 +129,40 @@ USER root
|
|||
# https://github.com/BerriAI/litellm/issues/33518
|
||||
RUN echo "https://packages.wolfi.dev/os" >> /etc/apk/repositories
|
||||
|
||||
# node (without npm) is required by the prisma CLI at runtime
|
||||
RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent
|
||||
# node (without npm) is required by the prisma CLI at runtime; nginx serves the
|
||||
# static admin UI in the `ui` component; libevent is pgbouncer's runtime.
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libatomic libevent nginx && break; \
|
||||
[ "$i" = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Runtime writes land under /app/.cache, /var/lib/litellm and /tmp (group-0
|
||||
# writable, so an arbitrary uid and a read-only root fs both work). Prisma CLI
|
||||
# and engines are baked under /opt/prisma so `prisma migrate deploy` needs no
|
||||
# npm and no network (#33650, #24554).
|
||||
ENV PATH="/app/.venv/bin:${PATH}" \
|
||||
PYTHONPATH="/app" \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
HOME=/app \
|
||||
LITELLM_NON_ROOT=true \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
|
||||
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
|
||||
PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
|
||||
PRISMA_HIDE_UPDATE_MESSAGE=1 \
|
||||
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
|
||||
PRISMA_OFFLINE_MODE=true
|
||||
|
||||
# Copy only what runtime needs. The application is installed inside the venv;
|
||||
# the rest of the builder's /app is source and build metadata that must not
|
||||
# ship (manifest-scanning tools attribute everything in it to this image).
|
||||
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
|
||||
COPY --from=builder /app/.venv /app/.venv
|
||||
COPY --from=builder /app/docker /app/docker
|
||||
COPY --from=builder /app/docker-entrypoint.sh /app/docker-entrypoint.sh
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
|
||||
# enterprise/ is imported by source path at runtime (proxy_cli puts the
|
||||
|
|
@ -149,21 +170,34 @@ COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/pr
|
|||
# enterprise.enterprise_hooks from it)
|
||||
COPY --from=builder /app/enterprise /app/enterprise
|
||||
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
|
||||
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every
|
||||
# runtime uid can read and that no cache volume mount shadows. The paths are
|
||||
# pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and recorded into the
|
||||
# generated client at build time, so `prisma migrate deploy` on a fresh
|
||||
# database needs no npm and no network access (#33650, #24554).
|
||||
COPY --from=builder /app/gateway /app/gateway
|
||||
COPY --from=builder /app/backend /app/backend
|
||||
COPY --from=builder /app/migrations /app/migrations
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
COPY --from=builder /var/lib/litellm /var/lib/litellm
|
||||
COPY ui/nginx.conf /etc/nginx/nginx.conf
|
||||
|
||||
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
|
||||
find /app/.venv -type d -path "*/tornado/test" -delete && \
|
||||
RUN find /app/.venv -depth -type d -path "*/tornado/test" -exec rm -rf {} + && \
|
||||
mkdir -p /app/.cache && \
|
||||
chown -R 65532:0 /app /var/lib/litellm && \
|
||||
chmod -R g=u,g+w /app/.cache /var/lib/litellm && \
|
||||
PRISMA_PATH="$(python -c 'import os, prisma; print(os.path.dirname(prisma.__file__))')" && \
|
||||
PROXY_EXTRAS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))')" && \
|
||||
chmod -R g=u,g+w "$PRISMA_PATH" "$PROXY_EXTRAS_PATH" && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
|
||||
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
|
||||
ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \
|
||||
python -c "from litellm.rust_bridge.loader import native_bridge_available; assert native_bridge_available()" && \
|
||||
python -c "import gateway.launch, backend.main"
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The
|
||||
# numeric form is what the kubelet's runAsNonRoot admission check can verify.
|
||||
USER 65532:65532
|
||||
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
CMD ["--port", "4000"]
|
||||
RUN nginx -t
|
||||
|
||||
EXPOSE 4000/tcp 4001/tcp 3000/tcp
|
||||
|
||||
ENTRYPOINT ["/app/docker-entrypoint.sh"]
|
||||
|
|
|
|||
|
|
@ -1,106 +0,0 @@
|
|||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
# ---------- Builder ----------
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
COPY --from=uvbin /uv /uvx /usr/local/bin/
|
||||
|
||||
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
|
||||
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
|
||||
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start.
|
||||
# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a
|
||||
# BuildKit cache mount (different filesystem).
|
||||
# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of
|
||||
# silently pulling a managed interpreter.
|
||||
# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't
|
||||
# silently re-enable nodeenv's Node download.
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
UV_COMPILE_BYTECODE=1 \
|
||||
UV_PYTHON_DOWNLOADS=0 \
|
||||
PRISMA_USE_GLOBAL_NODE=true \
|
||||
PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Stage 1 — install dependencies only.
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
|
||||
--mount=type=bind,source=uv.lock,target=uv.lock \
|
||||
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
|
||||
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
|
||||
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--python python3.13
|
||||
|
||||
# Stage 2 — copy source and install the project + workspace members.
|
||||
COPY . .
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--python python3.13
|
||||
|
||||
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh
|
||||
|
||||
# ---------- Runtime ----------
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
USER root
|
||||
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with
|
||||
# /home/nonroot. We run the backend as that user
|
||||
WORKDIR /app
|
||||
ENV HOME=/home/nonroot \
|
||||
PATH="/app/.venv/bin:${PATH}" \
|
||||
PYTHONPATH="/app" \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
|
||||
|
||||
COPY --from=builder --chown=nonroot:nonroot /app /app
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
|
||||
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
|
||||
find /app/.venv -type d -path "*/tornado/test" -delete && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
|
||||
|
||||
USER nonroot
|
||||
|
||||
EXPOSE 4001/tcp
|
||||
|
||||
ENTRYPOINT ["/app/docker/component_entrypoint.sh", "uvicorn", "backend.main:app"]
|
||||
CMD ["--host", "0.0.0.0", "--port", "4001"]
|
||||
|
|
@ -4,6 +4,5 @@
|
|||
|
||||
- Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Grype scan on `Dockerfile.database` (fails on CRITICAL)
|
||||
- Grype scan on main `Dockerfile` (fails on CRITICAL)
|
||||
- Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0)
|
||||
|
|
|
|||
|
|
@ -1,25 +0,0 @@
|
|||
FROM ollama/ollama as ollama
|
||||
|
||||
RUN echo "auto installing llama2"
|
||||
|
||||
# auto install ollama/llama2
|
||||
RUN ollama serve & sleep 2 && ollama pull llama2
|
||||
|
||||
RUN echo "installing litellm"
|
||||
|
||||
RUN apt-get update
|
||||
|
||||
# Install Python
|
||||
RUN apt-get install -y python3 python3-pip
|
||||
|
||||
# Set the working directory in the container
|
||||
WORKDIR /app
|
||||
|
||||
# Copy the current directory contents into the container at /app
|
||||
COPY . /app
|
||||
|
||||
# Install any needed packages specified in requirements.txt
|
||||
|
||||
RUN python3 -m pip install litellm
|
||||
COPY start.sh /start.sh
|
||||
ENTRYPOINT [ "/bin/bash", "/start.sh" ]
|
||||
|
|
@ -1 +0,0 @@
|
|||
litellm==1.83.14
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
ollama serve &
|
||||
litellm
|
||||
|
|
@ -1,35 +0,0 @@
|
|||
import openai
|
||||
|
||||
api_base = "http://0.0.0.0:8000"
|
||||
|
||||
openai.api_base = api_base
|
||||
openai.api_key = "temp-key"
|
||||
print(openai.api_base)
|
||||
|
||||
|
||||
print("LiteLLM: response from proxy with streaming")
|
||||
response = openai.ChatCompletion.create(
|
||||
model="ollama/llama2",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, acknowledge that you got it",
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(f"LiteLLM: streaming response from proxy {chunk}")
|
||||
|
||||
response = openai.ChatCompletion.create(
|
||||
model="ollama/llama2",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, acknowledge that you got it",
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(f"LiteLLM: response from proxy {response}")
|
||||
|
|
@ -1,46 +0,0 @@
|
|||
services:
|
||||
# Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints.
|
||||
# Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage.
|
||||
litellm:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile.non_root
|
||||
target: runtime
|
||||
args:
|
||||
PROXY_EXTRAS_SOURCE: "local"
|
||||
depends_on:
|
||||
- squid
|
||||
user: "101:101"
|
||||
group_add:
|
||||
- "2345"
|
||||
read_only: true
|
||||
cap_drop:
|
||||
- ALL
|
||||
security_opt:
|
||||
- no-new-privileges:true
|
||||
tmpfs:
|
||||
- /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777
|
||||
- /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777
|
||||
volumes:
|
||||
- ./proxy_server_config.yaml:/app/config.yaml:ro
|
||||
environment:
|
||||
LITELLM_NON_ROOT: "true"
|
||||
PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries"
|
||||
XDG_CACHE_HOME: "/app/cache"
|
||||
LITELLM_MIGRATION_DIR: "/app/migrations"
|
||||
HTTP_PROXY: "http://squid:3128"
|
||||
HTTPS_PROXY: "http://squid:3128"
|
||||
NO_PROXY: "localhost,127.0.0.1,db"
|
||||
command:
|
||||
- "--port"
|
||||
- "4000"
|
||||
- "--config"
|
||||
- "/app/config.yaml"
|
||||
squid:
|
||||
image: sameersbn/squid:3.5.27-2
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "3128:3128"
|
||||
tmpfs:
|
||||
- /var/spool/squid:rw,noexec,nosuid,nodev,size=64m
|
||||
- /var/log/squid:rw,noexec,nosuid,nodev,size=16m
|
||||
|
|
@ -1,52 +1,69 @@
|
|||
# LiteLLM plus a Postgres database that stores models, virtual keys and spend
|
||||
# logs. Used by https://docs.litellm.ai/docs/proxy/docker_quick_start
|
||||
#
|
||||
# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml
|
||||
# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
|
||||
# docker compose up -d
|
||||
#
|
||||
# `docker compose up` runs the published image. From a checkout of the repo,
|
||||
# `docker compose up --build` builds the Dockerfile next to this file instead.
|
||||
# `docker compose --profile monitoring up -d` also starts Prometheus on 9090.
|
||||
#
|
||||
# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY
|
||||
# makes credentials already stored in the database unreadable. For anything
|
||||
# beyond local evaluation, pin the image to a specific release tag.
|
||||
services:
|
||||
litellm:
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
build:
|
||||
context: .
|
||||
args:
|
||||
target: runtime
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
#########################################
|
||||
## Uncomment these lines to start proxy with a config.yaml file ##
|
||||
# volumes:
|
||||
# - ./config.yaml:/app/config.yaml
|
||||
# command:
|
||||
# - "--config=/app/config.yaml"
|
||||
##############################################
|
||||
target: runtime
|
||||
pull_policy: missing
|
||||
ports:
|
||||
- "4000:4000" # Map the container port to the host, change the host port if necessary
|
||||
- "4000:4000"
|
||||
# The same image runs every component. The first word of `command` picks
|
||||
# it: proxy (default), gateway, backend, ui, migrations, metrics,
|
||||
# collector. See docker/README.md
|
||||
# command: ["--config", "/app/config.yaml"]
|
||||
# volumes:
|
||||
# - ./config.yaml:/app/config.yaml:ro
|
||||
environment:
|
||||
DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm"
|
||||
# Optional: route read-only queries (find_*, count, group_by, query_raw/_first)
|
||||
# to a separate reader endpoint, e.g. an Aurora reader. Leave unset for
|
||||
# single-DB deployments. With IAM_TOKEN_DB_AUTH enabled, the reader URL
|
||||
# is auto-refreshed alongside the writer.
|
||||
# DATABASE_URL_READ_REPLICA: "postgresql://llmproxy:dbpassword9090@db-reader:5432/litellm"
|
||||
STORE_MODEL_IN_DB: "True" # allows adding models to proxy via UI
|
||||
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env, see the header of this file}
|
||||
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env, see the header of this file}
|
||||
DATABASE_URL: postgresql://llmproxy:dbpassword9090@db:5432/litellm
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
env_file:
|
||||
- .env # Load local .env file
|
||||
- path: .env
|
||||
required: false
|
||||
read_only: true
|
||||
cap_drop:
|
||||
- ALL
|
||||
security_opt:
|
||||
- no-new-privileges:true
|
||||
tmpfs:
|
||||
- /tmp:rw,noexec,nosuid,nodev,size=64m,mode=1777
|
||||
- /app/.cache:rw,noexec,nosuid,nodev,size=128m,mode=1777
|
||||
depends_on:
|
||||
- db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
|
||||
healthcheck: # Defines the health check configuration for the container
|
||||
db:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test:
|
||||
- CMD-SHELL
|
||||
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')" # Command to execute for health check
|
||||
interval: 30s # Perform health check every 30 seconds
|
||||
timeout: 10s # Health check command times out after 10 seconds
|
||||
retries: 3 # Retry up to 3 times if health check fails
|
||||
start_period: 40s # Wait 40 seconds after container start before beginning health checks
|
||||
- python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')"
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 40s
|
||||
|
||||
db:
|
||||
image: postgres:16
|
||||
restart: always
|
||||
container_name: litellm_db
|
||||
environment:
|
||||
POSTGRES_DB: litellm
|
||||
POSTGRES_USER: llmproxy
|
||||
POSTGRES_PASSWORD: dbpassword9090
|
||||
ports:
|
||||
- "5432:5432"
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data # Persists Postgres data across container restarts
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -d litellm -U llmproxy"]
|
||||
interval: 1s
|
||||
|
|
@ -55,6 +72,8 @@ services:
|
|||
|
||||
prometheus:
|
||||
image: prom/prometheus
|
||||
profiles:
|
||||
- monitoring
|
||||
volumes:
|
||||
- prometheus_data:/prometheus
|
||||
- ./prometheus.yml:/etc/prometheus/prometheus.yml
|
||||
|
|
@ -68,6 +87,5 @@ services:
|
|||
|
||||
volumes:
|
||||
prometheus_data:
|
||||
driver: local
|
||||
postgres_data:
|
||||
name: litellm_postgres_data # Named volume for Postgres data persistence
|
||||
name: litellm_postgres_data
|
||||
|
|
|
|||
93
docker-entrypoint.sh
Executable file
93
docker-entrypoint.sh
Executable file
|
|
@ -0,0 +1,93 @@
|
|||
#!/bin/sh
|
||||
# LiteLLM image entrypoint: docker-entrypoint.sh [COMPONENT] [ARGS...]
|
||||
#
|
||||
# COMPONENT selects the process this container runs; it can also be given as
|
||||
# LITELLM_COMPONENT when the command carries no component. Anything after it is
|
||||
# passed to that process. A first argument starting with "-" (or no argument at
|
||||
# all) keeps the historical behaviour of running the monolithic proxy.
|
||||
#
|
||||
# proxy litellm ARGS everything in one process (default)
|
||||
# gateway python -m gateway.launch ARGS inference routes, 0.0.0.0:4000
|
||||
# backend uvicorn backend.main:app ARGS management routes, 0.0.0.0:4001
|
||||
# ui nginx serving the admin UI port 3000
|
||||
# migrations python migrations/run.py prisma migrate deploy, then exit
|
||||
# metrics python -m litellm.proxy.prometheus_metrics_server ARGS
|
||||
# collector python -m litellm.proxy.collector ARGS
|
||||
#
|
||||
# gateway and backend get their host and port defaults first, so ARGS such as
|
||||
# --port 8080 override them. An unknown first word is executed as-is (docker run
|
||||
# <image> sh). PgBouncer is not a component: LITELLM_PGBOUNCER_ENABLED=true
|
||||
# starts it inside proxy and gateway. Components that write Prometheus samples
|
||||
# start with an empty PROMETHEUS_MULTIPROC_DIR; metrics, collector and raw
|
||||
# commands get the directory created but keep the workers' files.
|
||||
set -eu
|
||||
|
||||
usage() {
|
||||
sed -n '2,24p' "$0" | sed 's/^# \{0,1\}//' >&2
|
||||
}
|
||||
|
||||
component="${LITELLM_COMPONENT:-proxy}"
|
||||
case "${1:-}" in
|
||||
proxy|gateway|backend|ui|migrations|metrics|collector)
|
||||
component="$1"
|
||||
shift
|
||||
;;
|
||||
""|-*)
|
||||
;;
|
||||
help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
[ -z "${PROMETHEUS_MULTIPROC_DIR:-}" ] || mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
|
||||
exec "$@"
|
||||
;;
|
||||
esac
|
||||
|
||||
case "$component" in
|
||||
proxy)
|
||||
set -- litellm "$@"
|
||||
;;
|
||||
gateway)
|
||||
set -- python -m gateway.launch --workers "${NUM_WORKERS:-1}" --host 0.0.0.0 --port 4000 "$@"
|
||||
;;
|
||||
backend)
|
||||
set -- uvicorn backend.main:app --host 0.0.0.0 --port 4001 "$@"
|
||||
;;
|
||||
ui)
|
||||
exec nginx -g 'daemon off;' "$@"
|
||||
;;
|
||||
migrations)
|
||||
set -- python /app/migrations/run.py "$@"
|
||||
;;
|
||||
metrics)
|
||||
set -- python -m litellm.proxy.prometheus_metrics_server "$@"
|
||||
;;
|
||||
collector)
|
||||
set -- python -m litellm.proxy.collector "$@"
|
||||
;;
|
||||
*)
|
||||
echo "docker-entrypoint.sh: unknown LITELLM_COMPONENT '$component'" >&2
|
||||
usage
|
||||
exit 64
|
||||
;;
|
||||
esac
|
||||
|
||||
if [ -n "${PROMETHEUS_MULTIPROC_DIR:-}" ]; then
|
||||
case "$component" in
|
||||
metrics|collector) mkdir -p "$PROMETHEUS_MULTIPROC_DIR" ;;
|
||||
*)
|
||||
mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
|
||||
rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
|
||||
case "${USE_DDTRACE:-}" in
|
||||
[Tt][Rr][Uu][Ee])
|
||||
export DD_TRACE_OPENAI_ENABLED="False"
|
||||
exec ddtrace-run "$@"
|
||||
;;
|
||||
esac
|
||||
|
||||
exec "$@"
|
||||
|
|
@ -1,162 +0,0 @@
|
|||
# syntax=docker/dockerfile:1.7
|
||||
|
||||
# Base image for building
|
||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
|
||||
# Runtime image
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
|
||||
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
|
||||
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
|
||||
ARG PGBOUNCER_VERSION=1.25.2
|
||||
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
|
||||
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
|
||||
ARG PGBOUNCER_VERSION
|
||||
ARG PGBOUNCER_SHA256
|
||||
USER root
|
||||
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
|
||||
WORKDIR /build
|
||||
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
|
||||
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
|
||||
tar xzf pgbouncer.tar.gz --strip-components=1 && \
|
||||
./configure --prefix=/usr/local --with-openssl=/usr && \
|
||||
make -j"$(nproc)" pgbouncer && \
|
||||
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
# Admin UI builder. Pinned to the build platform so the architecture-independent
|
||||
# Next.js static export compiles once natively even in a multi-arch build,
|
||||
# instead of once per target arch under QEMU.
|
||||
FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder
|
||||
|
||||
ENV NEXT_TELEMETRY_DISABLED=1 \
|
||||
npm_config_fund=false \
|
||||
npm_config_audit=false
|
||||
|
||||
WORKDIR /ui
|
||||
|
||||
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
|
||||
RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline
|
||||
|
||||
COPY ui/litellm-dashboard/ ./
|
||||
RUN npm run build
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
COPY --from=uvbin /uv /usr/local/bin/uv
|
||||
COPY --from=uvbin /uvx /usr/local/bin/uvx
|
||||
|
||||
RUN apk add --no-cache \
|
||||
bash \
|
||||
gcc \
|
||||
python-3.13 \
|
||||
python-3.13-dev \
|
||||
openssl \
|
||||
openssl-dev \
|
||||
nodejs \
|
||||
npm \
|
||||
libsndfile
|
||||
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
UV_PYTHON_DOWNLOADS=0 \
|
||||
PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Copy dependency metadata first for layer caching
|
||||
COPY pyproject.toml uv.lock ./
|
||||
COPY enterprise/pyproject.toml enterprise/
|
||||
COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/
|
||||
|
||||
# Install third-party dependencies (cached unless pyproject.toml/uv.lock change)
|
||||
RUN uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
# Copy full source tree
|
||||
COPY . .
|
||||
|
||||
# Replace the committed UI bundle with the one built from this exact source.
|
||||
# Clearing first drops the committed bundle's content-hashed chunks that COPY
|
||||
# would otherwise leave behind alongside the fresh ones.
|
||||
RUN rm -rf litellm/proxy/_experimental/out
|
||||
COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/
|
||||
|
||||
# Build Admin UI before final sync (applies the enterprise color override when present)
|
||||
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
# Install project and workspace packages (fast - deps already cached)
|
||||
RUN uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
|
||||
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
USER root
|
||||
|
||||
# node (without npm) is required by the prisma CLI at runtime
|
||||
RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent
|
||||
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
WORKDIR /app
|
||||
ENV PATH="/app/.venv/bin:${PATH}" \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
|
||||
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
|
||||
PRISMA_OFFLINE_MODE=true
|
||||
|
||||
# Copy only what runtime needs. The application is installed inside the venv;
|
||||
# the rest of the builder's /app is source and build metadata that must not
|
||||
# ship (manifest-scanning tools attribute everything in it to this image).
|
||||
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
|
||||
COPY --from=builder /app/.venv /app/.venv
|
||||
COPY --from=builder /app/docker /app/docker
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
|
||||
# enterprise/ is imported by source path at runtime (proxy_cli puts the
|
||||
# working directory on sys.path; litellm/proxy/hooks resolves
|
||||
# enterprise.enterprise_hooks from it)
|
||||
COPY --from=builder /app/enterprise /app/enterprise
|
||||
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
|
||||
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every
|
||||
# runtime uid can read and that no cache volume mount shadows (unlike
|
||||
# /app/.cache or $HOME/.cache in readOnlyRootFilesystem + emptyDir setups).
|
||||
# The paths are pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and
|
||||
# recorded into the generated client at build time, so `prisma migrate
|
||||
# deploy` on a fresh database needs no npm and no network access
|
||||
# (#33650, #24554).
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
|
||||
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
|
||||
find /app/.venv -type d -path "*/tornado/test" -delete && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
|
||||
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
CMD ["--port", "4000"]
|
||||
|
|
@ -1,217 +0,0 @@
|
|||
# syntax=docker/dockerfile:1.7
|
||||
|
||||
# Base images
|
||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG PROXY_EXTRAS_SOURCE=published
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
|
||||
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
|
||||
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
|
||||
ARG PGBOUNCER_VERSION=1.25.2
|
||||
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
|
||||
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
|
||||
ARG PGBOUNCER_VERSION
|
||||
ARG PGBOUNCER_SHA256
|
||||
USER root
|
||||
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
|
||||
WORKDIR /build
|
||||
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
|
||||
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
|
||||
tar xzf pgbouncer.tar.gz --strip-components=1 && \
|
||||
./configure --prefix=/usr/local --with-openssl=/usr && \
|
||||
make -j"$(nproc)" pgbouncer && \
|
||||
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
# Admin UI builder. Pinned to the build platform so the architecture-independent
|
||||
# Next.js static export compiles once natively even in a multi-arch build,
|
||||
# instead of once per target arch under QEMU.
|
||||
FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder
|
||||
|
||||
ENV NEXT_TELEMETRY_DISABLED=1 \
|
||||
npm_config_fund=false \
|
||||
npm_config_audit=false
|
||||
|
||||
WORKDIR /ui
|
||||
|
||||
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
|
||||
RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline
|
||||
|
||||
COPY ui/litellm-dashboard/ ./
|
||||
RUN npm run build
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
COPY --from=uvbin /uv /usr/local/bin/uv
|
||||
COPY --from=uvbin /uvx /usr/local/bin/uvx
|
||||
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache \
|
||||
python-3.13 \
|
||||
python-3.13-dev \
|
||||
gcc \
|
||||
rust \
|
||||
bash \
|
||||
coreutils \
|
||||
curl \
|
||||
openssl \
|
||||
libsndfile \
|
||||
nodejs \
|
||||
npm && break || sleep 5; \
|
||||
done
|
||||
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
UV_PYTHON_DOWNLOADS=0 \
|
||||
PATH="/app/.venv/bin:${PATH}" \
|
||||
LITELLM_NON_ROOT=true \
|
||||
XDG_CACHE_HOME=/app/.cache
|
||||
|
||||
# Copy dependency metadata first for layer caching
|
||||
COPY pyproject.toml uv.lock ./
|
||||
COPY enterprise/pyproject.toml enterprise/
|
||||
COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/
|
||||
|
||||
# Install third-party dependencies (cached unless pyproject.toml/uv.lock change)
|
||||
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
|
||||
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
# Copy full source tree
|
||||
COPY . .
|
||||
|
||||
# Replace the committed UI bundle with the one built from this exact source.
|
||||
# Clearing first drops the committed bundle's content-hashed chunks that COPY
|
||||
# would otherwise leave behind alongside the fresh ones.
|
||||
RUN rm -rf litellm/proxy/_experimental/out
|
||||
COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/
|
||||
|
||||
# Set non-root flag for build time consistency
|
||||
ENV LITELLM_NON_ROOT=true
|
||||
|
||||
RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
cp -r /app/litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \
|
||||
cp /app/litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \
|
||||
touch /var/lib/litellm/ui/.litellm_ui_ready
|
||||
|
||||
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
|
||||
if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13 \
|
||||
--no-sources-package litellm-proxy-extras; \
|
||||
else \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13; \
|
||||
fi
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
|
||||
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
RUN for i in 1 2 3; do \
|
||||
apk upgrade --no-cache && break || sleep 5; \
|
||||
done && \
|
||||
for i in 1 2 3; do \
|
||||
apk add --no-cache python-3.13 bash openssl tzdata libsndfile nodejs libevent && break || sleep 5; \
|
||||
done
|
||||
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
# Copy only what runtime needs. The application is installed inside the venv;
|
||||
# the rest of the builder's /app is source and build metadata that must not
|
||||
# ship (manifest-scanning tools attribute everything in it to this image).
|
||||
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
|
||||
COPY --from=builder /app/.venv /app/.venv
|
||||
COPY --from=builder /app/docker /app/docker
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
|
||||
# enterprise/ is imported by source path at runtime (proxy_cli puts the
|
||||
# working directory on sys.path; litellm/proxy/hooks resolves
|
||||
# enterprise.enterprise_hooks from it)
|
||||
COPY --from=builder /app/enterprise /app/enterprise
|
||||
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
|
||||
# Prisma CLI + engines are baked under /opt/prisma, a fixed path every runtime
|
||||
# uid can read and that no cache volume mount shadows (unlike /app/.cache or
|
||||
# $HOME/.cache under readOnlyRootFilesystem + emptyDir or arbitrary-uid setups).
|
||||
# PRISMA_CLI_QUERY_ENGINE_TYPE=binary makes the CLI use the baked binary query
|
||||
# engine directly, so `prisma migrate deploy` on a fresh database needs no npm
|
||||
# and no network access; without it the CLI looks for the library engine, which
|
||||
# prisma stopped baking, and falls back to a download that fails offline or as a
|
||||
# non-writable uid (#33650, #24554).
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
COPY --from=builder /var/lib/litellm/ui /var/lib/litellm/ui
|
||||
COPY --from=builder /var/lib/litellm/assets /var/lib/litellm/assets
|
||||
|
||||
# XDG_CACHE_HOME is intentionally left unset so it falls back to $HOME/.cache
|
||||
# (/app/.cache, writable by the runtime uid). The prisma bake at the read-only
|
||||
# /opt/prisma is anchored by PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH, so
|
||||
# nothing needs XDG to point there; pointing it at the read-only bake would
|
||||
# deny any XDG-aware library that writes a cache at runtime.
|
||||
ENV PATH="/app/.venv/bin:${PATH}" \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
|
||||
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
|
||||
HOME=/app \
|
||||
LITELLM_NON_ROOT=true \
|
||||
PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
|
||||
PRISMA_HIDE_UPDATE_MESSAGE=1 \
|
||||
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
|
||||
PRISMA_OFFLINE_MODE=true
|
||||
|
||||
RUN mkdir -p /nonexistent /app/.cache /var/lib/litellm/assets /var/lib/litellm/ui && \
|
||||
chown -R nobody:nogroup /app /var/lib/litellm/ui /var/lib/litellm/assets /nonexistent && \
|
||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
chown -R nobody:nogroup "$PRISMA_PATH" && \
|
||||
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
||||
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup "$LITELLM_PKG_MIGRATIONS_PATH" || true && \
|
||||
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
||||
chgrp -R 0 "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 "$LITELLM_PROXY_EXTRAS_PATH" || true && \
|
||||
chmod -R g=u "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u "$LITELLM_PROXY_EXTRAS_PATH" || true && \
|
||||
chmod -R g+w "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w "$LITELLM_PROXY_EXTRAS_PATH" || true && \
|
||||
chmod -R g+rX "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
|
||||
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \
|
||||
ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null 2>&1 && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths"
|
||||
|
||||
USER 65534
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||
CMD ["--port", "4000"]
|
||||
|
|
@ -2,15 +2,15 @@
|
|||
|
||||
This guide provides instructions for building and running the LiteLLM application using Docker and Docker Compose.
|
||||
|
||||
> **Just want to run LiteLLM?** This guide builds from source. To run the published
|
||||
> image instead, use `docker-compose.quickstart.yml` in this directory — the
|
||||
> two-service stack (gateway + Postgres) that the
|
||||
> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents:
|
||||
> **Just want to run LiteLLM?** `docker-compose.yml` in the repository root runs
|
||||
> the published image with a Postgres database, the stack the
|
||||
> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents,
|
||||
> and it works on its own outside a checkout:
|
||||
>
|
||||
> ```bash
|
||||
> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml
|
||||
> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml
|
||||
> printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
|
||||
> docker compose -f docker-compose.quickstart.yml up -d
|
||||
> docker compose up -d
|
||||
> ```
|
||||
|
||||
## Prerequisites
|
||||
|
|
@ -20,33 +20,52 @@ This guide provides instructions for building and running the LiteLLM applicatio
|
|||
|
||||
## Building and Running the Application
|
||||
|
||||
To build and run the application, you will use the `docker-compose.yml` file located in the root of the project. This file is configured to use the `Dockerfile.non_root` for a secure, non-root container environment.
|
||||
The same `docker-compose.yml` builds from source when you pass `--build`: it builds the `Dockerfile` in the repository root, the one image LiteLLM ships. `docker-entrypoint.sh` next to it is the only entrypoint script; every container starts through it
|
||||
|
||||
### 1. Set the Master Key
|
||||
## One image, many components
|
||||
|
||||
The application requires a `LITELLM_MASTER_KEY` for signing and validating tokens. You must set this key as an environment variable before running the application.
|
||||
Every LiteLLM container runs the same image. The first word of the container command (or the `LITELLM_COMPONENT` environment variable when the command carries only flags) picks the process the container runs, and everything after it is handed to that process unchanged:
|
||||
|
||||
Create a `.env` file in the root of the project and add the following line:
|
||||
| Component | Runs | Port |
|
||||
|--------------|-----------------------------------------------------|------|
|
||||
| `proxy` | `litellm ...` (everything in one process, default) | 4000 |
|
||||
| `gateway` | `python -m gateway.launch ...` (inference routes) | 4000 |
|
||||
| `backend` | `uvicorn backend.main:app ...` (management routes) | 4001 |
|
||||
| `ui` | nginx serving the static admin UI | 3000 |
|
||||
| `migrations` | `python migrations/run.py`, `prisma migrate deploy` then exit | |
|
||||
| `metrics` | `python -m litellm.proxy.prometheus_metrics_server ...` | `--port` |
|
||||
| `collector` | `python -m litellm.proxy.collector ...` | |
|
||||
|
||||
```
|
||||
LITELLM_MASTER_KEY=your-secret-key
|
||||
```bash
|
||||
docker run -p 4000:4000 litellm --config /app/config.yaml # proxy, exactly as before
|
||||
docker run -p 4000:4000 litellm gateway --port 4000 # componentized data plane
|
||||
docker run -p 4001:4001 -e LITELLM_COMPONENT=backend litellm # same, chosen through the env
|
||||
docker run -p 3000:3000 --read-only --tmpfs /tmp litellm ui # admin UI behind nginx
|
||||
docker run -e DATABASE_URL=... litellm migrations # one-off schema migration job
|
||||
docker run -it litellm sh # anything else runs verbatim
|
||||
```
|
||||
|
||||
Replace `your-secret-key` with a strong, randomly generated secret.
|
||||
PgBouncer is not a separate component: `LITELLM_PGBOUNCER_ENABLED=true` starts an in-container PgBouncer in front of `DATABASE_URL` inside `proxy` and `gateway`. `USE_DDTRACE=true` wraps whichever component runs with `ddtrace-run`, and `PROMETHEUS_MULTIPROC_DIR` is emptied of stale samples before any workers fork (the `metrics` and `collector` sidecars only read it, so their restart keeps the live samples)
|
||||
|
||||
The image runs as uid `65532` (`nonroot` in the Wolfi base) and also works as an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, because everything it writes at runtime lives under `/app/.cache`, `/var/lib/litellm` and `/tmp`. Mount those (or set `readOnlyRootFilesystem` with emptyDirs there) for a read-only root filesystem. Prisma's CLI and engines are baked under `/opt/prisma`, so migrations need neither network nor a writable home
|
||||
|
||||
### 1. Set the Master and Salt Keys
|
||||
|
||||
The proxy signs virtual keys with `LITELLM_MASTER_KEY` and encrypts stored provider credentials with `LITELLM_SALT_KEY`. Compose reads both from a `.env` file in the directory you run it from, so generate them once:
|
||||
|
||||
```bash
|
||||
printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
|
||||
```
|
||||
|
||||
Keep the file: regenerating `LITELLM_SALT_KEY` makes credentials already stored in the database unreadable. Provider keys such as `OPENAI_API_KEY` go in the same file, and the whole file is passed to the container
|
||||
|
||||
### 2. Build and Run the Containers
|
||||
|
||||
Once you have set the `LITELLM_MASTER_KEY`, you can build and run the containers using the following command:
|
||||
|
||||
```bash
|
||||
docker compose up -d --build
|
||||
```
|
||||
|
||||
This command will:
|
||||
|
||||
- Build the Docker image using `Dockerfile.non_root`.
|
||||
- Start the `litellm`, `litellm_db`, and `prometheus` services in detached mode (`-d`).
|
||||
- The `--build` flag ensures that the image is rebuilt if there are any changes to the Dockerfile or the application code.
|
||||
This command builds the image from the root `Dockerfile` and starts the `litellm` and `db` services in detached mode. Without `--build`, `docker compose up` pulls the published `main-stable` image instead. Add `--profile monitoring` to also start Prometheus on port 9090, scraping the proxy with the root `prometheus.yml`
|
||||
|
||||
### 3. Verifying the Application is Running
|
||||
|
||||
|
|
@ -70,34 +89,18 @@ To stop the running containers, use the following command:
|
|||
docker compose down
|
||||
```
|
||||
|
||||
## Hardened / Offline Testing
|
||||
## Hardening
|
||||
|
||||
To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file:
|
||||
The compose file runs the proxy the way a locked-down cluster would: as the image's non-root user with a read-only root filesystem, every capability dropped, `no-new-privileges` set, and tmpfs mounts only at `/tmp` and `/app/.cache`. The image is built for that, so a change that makes the proxy write anywhere else fails here before it fails in Kubernetes. To try an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, add `user: "101:0"` to the `litellm` service
|
||||
|
||||
Prisma's CLI and engines are baked under `/opt/prisma`, so migrations run without network access. Verify with:
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache
|
||||
docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d
|
||||
docker run --rm --network none --entrypoint prisma docker.litellm.ai/berriai/litellm:main-stable --version
|
||||
```
|
||||
|
||||
This setup:
|
||||
- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image.
|
||||
- Runs the proxy as a non-root user with a read-only rootfs and only writable tmpfs mounts:
|
||||
- `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`)
|
||||
- `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`)
|
||||
- Pre-builds and serves the admin UI from read-only paths:
|
||||
- `/var/lib/litellm/ui` (pre-restructured Next.js UI with `.litellm_ui_ready` marker)
|
||||
- `/var/lib/litellm/assets` (UI logos and assets)
|
||||
- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines.
|
||||
|
||||
You should also verify offline Prisma behaviour with:
|
||||
|
||||
```bash
|
||||
docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version
|
||||
```
|
||||
|
||||
This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
|
||||
- **`Master key is not initialized`**: This error means the `LITELLM_MASTER_KEY` environment variable is not set. Make sure you have created a `.env` file in the project root with the `LITELLM_MASTER_KEY` defined.
|
||||
- **`required variable LITELLM_MASTER_KEY is missing a value`**: Compose did not find a `.env` file with `LITELLM_MASTER_KEY` and `LITELLM_SALT_KEY` in the directory you ran it from. Generate one as shown above.
|
||||
- **`password authentication failed for user "llmproxy"`**: the `litellm_postgres_data` volume was initialised by an older compose file with different credentials. `docker compose down -v` drops it and the next `up` recreates the database.
|
||||
- **`build_admin_ui.sh: not found`**: the build context is wrong. Run `docker compose` from the root of the repository.
|
||||
|
|
|
|||
|
|
@ -1,64 +0,0 @@
|
|||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.10.9@sha256:10902f58a1606787602f303954cea099626a4adb02acbac4c69920fe9d278f82
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
FROM python:3.13-slim@sha256:739e7213785e88c0f702dcdc12c0973afcbd606dbf021a589cab77d6b00b579d
|
||||
|
||||
ARG LITELLM_VERSION=1.83.0
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=uvbin /uv /usr/local/bin/uv
|
||||
COPY --from=uvbin /uvx /usr/local/bin/uvx
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends gcc libffi-dev nodejs npm && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
COPY schema.prisma .
|
||||
|
||||
# This image is specifically for validating/installing the published PyPI
|
||||
# artifact, not the checked-out source tree.
|
||||
# Keep the moved proxy-runtime packages explicit until the published PyPI
|
||||
# artifact includes that extra; newer releases will simply dedupe these.
|
||||
RUN uv venv --python python && \
|
||||
uv pip install --python /app/.venv/bin/python \
|
||||
"litellm[proxy,proxy-runtime]==${LITELLM_VERSION}" \
|
||||
"google-cloud-aiplatform==1.133.0" \
|
||||
"google-genai==1.37.0" \
|
||||
"anthropic[vertex]==0.84.0" \
|
||||
"grpcio==1.78.0" \
|
||||
"prometheus-client==0.20.0" \
|
||||
"langfuse==2.59.7" \
|
||||
"opentelemetry-api==1.28.0" \
|
||||
"opentelemetry-sdk==1.28.0" \
|
||||
"opentelemetry-exporter-otlp==1.28.0" \
|
||||
"ddtrace==4.11.0" \
|
||||
"sentry-sdk==2.21.0" \
|
||||
"mangum==0.17.0" \
|
||||
"azure-ai-contentsafety==1.0.0" \
|
||||
"azure-storage-file-datalake==12.20.0" \
|
||||
"pypdf==6.7.5" \
|
||||
"llm-sandbox==0.3.31" \
|
||||
"detect-secrets==1.5.0" \
|
||||
"prisma==0.11.0" \
|
||||
"openai==2.24.0"
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
python -c "import sys; from prisma.client import BINARY_PATHS; bad = sorted(p for group in BINARY_PATHS.model_dump().values() for p in group.values() if not p.startswith('/opt/prisma/')); sys.exit('prisma engines baked outside /opt/prisma: %r' % bad) if bad else None"
|
||||
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
|
||||
|
||||
COPY docker/prod_entrypoint.sh /app/docker/prod_entrypoint.sh
|
||||
RUN sed -i 's/\r$//' /app/docker/prod_entrypoint.sh && chmod +x /app/docker/prod_entrypoint.sh
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||
CMD ["--port", "4000"]
|
||||
|
|
@ -1,9 +0,0 @@
|
|||
# Docker to build LiteLLM Proxy from litellm pip package
|
||||
|
||||
### When to use this ?
|
||||
|
||||
If you need to build LiteLLM Proxy from litellm pip package, you can use this Dockerfile as a reference.
|
||||
|
||||
### Why build from pip package ?
|
||||
|
||||
- If your company has a strict requirement around security / building images you can follow steps outlined here
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
#!/bin/sh
|
||||
|
||||
# stale samples from a previous container incarnation would be summed into the aggregate
|
||||
if [ -n "$PROMETHEUS_MULTIPROC_DIR" ]; then
|
||||
mkdir -p "$PROMETHEUS_MULTIPROC_DIR"
|
||||
rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db
|
||||
fi
|
||||
|
||||
case "$USE_DDTRACE" in
|
||||
[Tt][Rr][Uu][Ee])
|
||||
export DD_TRACE_OPENAI_ENABLED="False"
|
||||
exec ddtrace-run "$@"
|
||||
;;
|
||||
esac
|
||||
|
||||
exec "$@"
|
||||
|
|
@ -1,41 +0,0 @@
|
|||
# LiteLLM quickstart stack: the gateway plus a Postgres database that stores
|
||||
# models, virtual keys, and spend logs. Used by
|
||||
# https://docs.litellm.ai/docs/proxy/docker_quick_start
|
||||
#
|
||||
# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml
|
||||
# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env
|
||||
# docker compose -f docker-compose.quickstart.yml up -d
|
||||
#
|
||||
# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY
|
||||
# makes credentials already stored in the database unreadable. For anything
|
||||
# beyond local evaluation, pin the image to a specific release tag.
|
||||
services:
|
||||
litellm:
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
ports:
|
||||
- "4000:4000"
|
||||
environment:
|
||||
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file}
|
||||
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file}
|
||||
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
|
||||
db:
|
||||
image: postgres:16
|
||||
environment:
|
||||
POSTGRES_USER: litellm
|
||||
POSTGRES_PASSWORD: litellm
|
||||
POSTGRES_DB: litellm
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U litellm"]
|
||||
interval: 5s
|
||||
timeout: 5s
|
||||
retries: 10
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
|
|
@ -1,16 +0,0 @@
|
|||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
VENV_PYTHON="$REPO_ROOT/.venv/bin/python"
|
||||
MIGRATION_SCRIPT="$REPO_ROOT/litellm/proxy/prisma_migration.py"
|
||||
|
||||
if [ -x "$VENV_PYTHON" ]; then
|
||||
"$VENV_PYTHON" "$MIGRATION_SCRIPT"
|
||||
elif command -v uv >/dev/null 2>&1; then
|
||||
(cd "$REPO_ROOT" && uv run --no-sync python "$MIGRATION_SCRIPT")
|
||||
else
|
||||
python3 "$MIGRATION_SCRIPT"
|
||||
fi
|
||||
|
||||
echo "Migration script ran successfully!"
|
||||
|
|
@ -1,4 +0,0 @@
|
|||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# semantic-router dependencies are installed via `uv sync`.
|
||||
|
|
@ -1,10 +0,0 @@
|
|||
#!/bin/sh
|
||||
|
||||
case "$USE_DDTRACE" in
|
||||
[Tt][Rr][Uu][Ee])
|
||||
export DD_TRACE_OPENAI_ENABLED="False"
|
||||
exec ddtrace-run litellm "$@"
|
||||
;;
|
||||
esac
|
||||
|
||||
exec litellm "$@"
|
||||
|
|
@ -1,18 +0,0 @@
|
|||
schemaVersion: 2.0.0
|
||||
|
||||
metadataTest:
|
||||
entrypoint: ["docker/prod_entrypoint.sh"]
|
||||
user: "65534"
|
||||
workdir: "/app"
|
||||
|
||||
fileExistenceTests:
|
||||
- name: "Prisma Folder"
|
||||
path: "/usr/local/lib/python3.13/site-packages/prisma/"
|
||||
shouldExist: true
|
||||
uid: 65534
|
||||
gid: 65534
|
||||
- name: "Prisma Schema"
|
||||
path: "/usr/local/lib/python3.13/site-packages/prisma/schema.prisma"
|
||||
shouldExist: true
|
||||
uid: 65534
|
||||
gid: 65534
|
||||
|
|
@ -1,126 +0,0 @@
|
|||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x)
|
||||
ARG PGBOUNCER_VERSION=1.25.2
|
||||
ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332
|
||||
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder
|
||||
ARG PGBOUNCER_VERSION
|
||||
ARG PGBOUNCER_SHA256
|
||||
USER root
|
||||
RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl
|
||||
WORKDIR /build
|
||||
RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \
|
||||
echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \
|
||||
tar xzf pgbouncer.tar.gz --strip-components=1 && \
|
||||
./configure --prefix=/usr/local --with-openssl=/usr && \
|
||||
make -j"$(nproc)" pgbouncer && \
|
||||
install -m 0755 pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
# ---------- Builder ----------
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
COPY --from=uvbin /uv /uvx /usr/local/bin/
|
||||
|
||||
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
|
||||
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
|
||||
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start.
|
||||
# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a
|
||||
# BuildKit cache mount (different filesystem).
|
||||
# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of
|
||||
# silently pulling a managed interpreter.
|
||||
# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't
|
||||
# silently re-enable nodeenv's Node download.
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
UV_COMPILE_BYTECODE=1 \
|
||||
UV_PYTHON_DOWNLOADS=0 \
|
||||
PRISMA_USE_GLOBAL_NODE=true \
|
||||
PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Stage 1 — install dependencies only.
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
|
||||
--mount=type=bind,source=uv.lock,target=uv.lock \
|
||||
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
|
||||
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
|
||||
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
# Stage 2 — copy source and install the project + workspace members.
|
||||
COPY . .
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
# PYTHONPATH=/app makes the source tree shadow the installed package, so the
|
||||
# compiled Rust extension must live next to the source or it is never imported.
|
||||
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh
|
||||
|
||||
# ---------- Runtime ----------
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
USER root
|
||||
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic libevent && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with
|
||||
# /home/nonroot. We run the proxy as that user.
|
||||
WORKDIR /app
|
||||
ENV HOME=/home/nonroot \
|
||||
PATH="/app/.venv/bin:${PATH}" \
|
||||
PYTHONPATH="/app" \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries
|
||||
|
||||
COPY --from=builder --chown=nonroot:nonroot /app /app
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer
|
||||
|
||||
RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \
|
||||
find /app/.venv -type d -path "*/tornado/test" -delete && \
|
||||
chmod -R a+rX /opt/prisma && \
|
||||
python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \
|
||||
python -c "import litellm; from litellm.rust_bridge.loader import native_bridge_available; assert litellm.__file__ == '/app/litellm/__init__.py', litellm.__file__; assert native_bridge_available()"
|
||||
|
||||
USER nonroot
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["sh", "-c", "exec /app/docker/component_entrypoint.sh python -m gateway.launch --workers \"${NUM_WORKERS:-1}\" \"$@\"", "--"]
|
||||
CMD ["--host", "0.0.0.0", "--port", "4000"]
|
||||
|
|
@ -1,2 +0,0 @@
|
|||
#!/bin/bash
|
||||
python3 proxy_cli.py
|
||||
|
|
@ -1,120 +0,0 @@
|
|||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
|
||||
FROM $UV_IMAGE AS uvbin
|
||||
|
||||
# ---------- Builder ----------
|
||||
#
|
||||
# Minimal install for `prisma migrate deploy`. We deliberately skip the heavy
|
||||
# `proxy-runtime` (otel, sentry, ddtrace, pypdf, google-genai, anthropic-vertex,
|
||||
# ...) and `semantic-router` extras that the gateway/backend pull in — the
|
||||
# migration engine doesn't need them. We DO install `--extra proxy` so the
|
||||
# DB-URL helper from `litellm.proxy.auth.rds_iam_token` is importable, which
|
||||
# is how the gateway and backend assemble `DATABASE_URL` at pod startup when
|
||||
# `IAM_TOKEN_DB_AUTH=true` (see backend/main.py:17, gateway/main.py:22). And
|
||||
# `--extra extra_proxy` provides the `prisma` CLI + the secret-manager
|
||||
# backends `litellm.secret_managers.main` lazily imports.
|
||||
#
|
||||
# `prisma generate` runs once at BUILD time to (a) install the Node-based
|
||||
# Prisma CLI into the binary cache and (b) download the migration / query
|
||||
# engine binaries. The Python client it also produces is unused by this
|
||||
# image's runtime entrypoint — that's fine, it's a few hundred KB and the
|
||||
# alternative (`prisma py fetch`) doesn't reliably trigger engine downloads
|
||||
# under nodeenv. Crucially we do NOT run `prisma generate` at RUNTIME; the
|
||||
# old migration job did, on every pod start, which is the wasteful behaviour
|
||||
# the componentization is fixing.
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
COPY --from=uvbin /uv /uvx /usr/local/bin/
|
||||
|
||||
# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE
|
||||
# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi
|
||||
# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes.
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
ENV UV_PROJECT_ENVIRONMENT=/app/.venv \
|
||||
UV_LINK_MODE=copy \
|
||||
UV_COMPILE_BYTECODE=1 \
|
||||
UV_PYTHON_DOWNLOADS=0 \
|
||||
PRISMA_USE_GLOBAL_NODE=true \
|
||||
PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Stage 1 — install third-party deps only (cached by pyproject.toml/uv.lock).
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
|
||||
--mount=type=bind,source=uv.lock,target=uv.lock \
|
||||
--mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \
|
||||
--mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \
|
||||
uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra extra_proxy \
|
||||
--python python3.13
|
||||
|
||||
# Stage 2 — copy source and install the project + workspace members.
|
||||
COPY . .
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra extra_proxy \
|
||||
--python python3.13
|
||||
|
||||
RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/
|
||||
|
||||
COPY migrations/run.py /app/run.py
|
||||
|
||||
# Pre-warm the Prisma binary cache so the Job pod doesn't reach the
|
||||
# internet on first start. This matches what the backend Dockerfile does:
|
||||
# `prisma generate` runs nodeenv (downloads Node), installs the prisma npm
|
||||
# CLI, downloads the engine binaries for each `binaryTarget` in
|
||||
# schema.prisma, AND emits the generated Python client. We don't need the
|
||||
# client at runtime — the migration job invokes `prisma migrate deploy`
|
||||
# via subprocess — but having it cached is harmless and the alternative
|
||||
# (`prisma py fetch`) doesn't reliably trigger engine downloads.
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
prisma generate --schema=./schema.prisma
|
||||
|
||||
# ---------- Runtime ----------
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
USER root
|
||||
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache bash openssl tzdata python-3.13 nodejs libsndfile libatomic && break; \
|
||||
[ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \
|
||||
sleep 5; \
|
||||
done
|
||||
|
||||
# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The
|
||||
# Prisma engine binaries are dynamically linked against libssl/libcrypto, so
|
||||
# openssl stays in the runtime layer.
|
||||
WORKDIR /app
|
||||
ENV HOME=/home/nonroot \
|
||||
PATH="/app/.venv/bin:${PATH}" \
|
||||
PYTHONPATH="/app" \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \
|
||||
PRISMA_CLI_QUERY_ENGINE_TYPE=binary \
|
||||
PRISMA_OFFLINE_MODE=true
|
||||
|
||||
COPY --from=builder --chown=nonroot:nonroot /app /app
|
||||
COPY --from=builder /opt/prisma /opt/prisma
|
||||
|
||||
RUN chmod -R a+rX /opt/prisma && \
|
||||
test -x /opt/prisma/binaries/node_modules/.bin/prisma && \
|
||||
test -f /opt/prisma/binaries/node_modules/prisma/build/index.js
|
||||
|
||||
USER nonroot
|
||||
|
||||
ENTRYPOINT ["python3", "/app/run.py"]
|
||||
|
|
@ -5,5 +5,5 @@ model_list:
|
|||
api_key: fake-key
|
||||
api_base: os.environ/FAKE_OPENAI_API_BASE
|
||||
|
||||
general_settings:
|
||||
alerting: ["slack"]
|
||||
general_settings:
|
||||
alerting: ["slack"]
|
||||
|
|
@ -1,34 +0,0 @@
|
|||
FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
|
||||
|
||||
ARG GH_VERSION=2.101.0
|
||||
ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8
|
||||
ARG UV_VERSION=0.10.9
|
||||
ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
|
||||
|
||||
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \
|
||||
&& echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \
|
||||
&& tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \
|
||||
&& rm /tmp/gh.tar.gz
|
||||
|
||||
RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \
|
||||
&& echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \
|
||||
&& tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \
|
||||
&& rm /tmp/uv.tar.gz
|
||||
|
||||
RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator
|
||||
|
||||
ENV HOME=/home/populator \
|
||||
LITELLM_REPO=/opt/litellm \
|
||||
DISABLE_AUTOUPDATER=1
|
||||
|
||||
COPY --chown=populator:populator . /opt/litellm/tests/e2e/
|
||||
|
||||
USER populator
|
||||
WORKDIR /home/populator
|
||||
CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"]
|
||||
|
|
@ -1,12 +1,12 @@
|
|||
# Render cron job for the Claude Code compatibility-matrix populator
|
||||
|
||||
The populator runs daily as the Render cron job `litellm-compat-matrix`
|
||||
(Docker runtime, built from the `Dockerfile` in this directory) rather
|
||||
(Docker runtime; the image recipe is in [Image](#image) below) rather
|
||||
than as a GitHub Action or on a dedicated VM. Trade-offs:
|
||||
|
||||
- ✅ No machine to keep on or patch. Render builds the image from this
|
||||
directory on every push to `main` that touches `tests/e2e/**` and
|
||||
runs it on the schedule.
|
||||
- ✅ No machine to keep on or patch. Render builds the image on every
|
||||
push to `main` that touches `tests/e2e/**` and runs it on the
|
||||
schedule.
|
||||
- ✅ Credentials live in Render env vars and secret files, scoped to
|
||||
this one service, instead of on a VM filesystem.
|
||||
- ✅ The publish token still uses the `mateo-berri` account, which is a
|
||||
|
|
@ -25,7 +25,6 @@ than as a GitHub Action or on a dedicated VM. Trade-offs:
|
|||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `Dockerfile` | The image Render builds: Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the non-root user `populator` (uid/gid 1000, which is what Render's secret files are readable by). |
|
||||
| `run_daily.sh` | The actual cron job. Resolves versions, clones the worktree, installs the Claude Code CLI under test, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. |
|
||||
| `install_claude_code.sh` | Downloads one Claude Code release (`<version> <dest-dir>`) from the vendor's native release channel, verifies it against the sha256 in that release's `manifest.json`, and refuses a binary whose `--version` disagrees. Run by the cron and by the `compat-matrix-image` GitHub workflow. |
|
||||
| `build_matrix.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.build_from_paths`. Exists only because the bash script needs *some* way to render the per-cell aggregation, and the builder is already Python. |
|
||||
|
|
@ -106,7 +105,7 @@ the same values if it ever has to be rebuilt.
|
|||
| Workspace | Litellm (the one that already builds the other litellm services) |
|
||||
| Type | Cron job, Docker runtime |
|
||||
| Repo / branch | `BerriAI/litellm` @ `main` |
|
||||
| Dockerfile path | `tests/e2e/claude_code/cron_vm/Dockerfile` |
|
||||
| Image | Built from the recipe under [Image](#image). Render only builds a Dockerfile that lives in the connected repo, and this repo ships exactly one Dockerfile (the LiteLLM image in its root), so the service has to point at a copy of the recipe kept with the service or at a prebuilt image |
|
||||
| Docker build context | `tests/e2e` (the repo root `.dockerignore` excludes `tests`, so the context has to start below it) |
|
||||
| Build filter | included paths `tests/e2e/**` |
|
||||
| Schedule | `0 6 * * *` (06:00 UTC daily) |
|
||||
|
|
@ -117,7 +116,7 @@ the same values if it ever has to be rebuilt.
|
|||
Render mounts secret files at `/etc/secrets/<name>`, which is where
|
||||
`CREDENTIALS_DIRECTORY` and `GOOGLE_APPLICATION_CREDENTIALS` in the env
|
||||
example point. Render also passes env vars to `docker build` as build
|
||||
args, which is why the `Dockerfile` declares no `ARG` that could ever
|
||||
args, which is why the image recipe declares no `ARG` that could ever
|
||||
be given a secret's name.
|
||||
|
||||
Creating it through the API looks like this (fill `envVars` and
|
||||
|
|
@ -145,13 +144,59 @@ curl -fsS https://api.render.com/v1/services \
|
|||
"plan": "4c-16g",
|
||||
"region": "oregon",
|
||||
"envSpecificDetails": {
|
||||
"dockerfilePath": "tests/e2e/claude_code/cron_vm/Dockerfile",
|
||||
"dockerfilePath": "<path to the recipe below in the repo the service builds from>",
|
||||
"dockerContext": "tests/e2e"
|
||||
}
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
## Image
|
||||
|
||||
Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with
|
||||
this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the
|
||||
non-root user `populator` (uid/gid 1000, which is what Render's secret
|
||||
files are readable by). This is the recipe the Render service builds;
|
||||
it lives with the service because the repo ships exactly one Dockerfile,
|
||||
the LiteLLM image in its root
|
||||
|
||||
```dockerfile
|
||||
FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251
|
||||
|
||||
ARG GH_VERSION=2.101.0
|
||||
ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8
|
||||
ARG UV_VERSION=0.10.9
|
||||
ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594
|
||||
|
||||
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \
|
||||
&& echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \
|
||||
&& tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \
|
||||
&& rm /tmp/gh.tar.gz
|
||||
|
||||
RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \
|
||||
&& echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \
|
||||
&& tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \
|
||||
&& rm /tmp/uv.tar.gz
|
||||
|
||||
RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator
|
||||
|
||||
ENV HOME=/home/populator \
|
||||
LITELLM_REPO=/opt/litellm \
|
||||
DISABLE_AUTOUPDATER=1
|
||||
|
||||
COPY --chown=populator:populator . /opt/litellm/tests/e2e/
|
||||
|
||||
USER populator
|
||||
WORKDIR /home/populator
|
||||
CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"]
|
||||
```
|
||||
|
||||
## Operating it
|
||||
|
||||
```bash
|
||||
|
|
@ -185,7 +230,7 @@ curl -fsS "https://api.render.com/v1/services/${CRON_ID}/deploys?limit=1" \
|
|||
# Build and run the image locally (docker on Apple silicon needs the
|
||||
# platform flag; the context is tests/e2e, see the table above).
|
||||
docker build --platform linux/amd64 \
|
||||
-f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix tests/e2e
|
||||
-f /path/to/compat-matrix.Dockerfile -t compat-matrix tests/e2e
|
||||
docker run --rm --platform linux/amd64 \
|
||||
--env-file litellm-compat-matrix.env -e SKIP_PUBLISH=1 \
|
||||
-v "$PWD/secrets:/etc/secrets:ro" compat-matrix
|
||||
|
|
@ -234,8 +279,8 @@ docker run --rm --platform linux/amd64 \
|
|||
A CLI release that breaks a cell shows up as a green→red flip, which
|
||||
withholds auto-merge on that day's docs PR for review. To rerun the
|
||||
matrix on one specific CLI, set `CLAUDE_CODE_VERSION` on the run.
|
||||
`gh` and `uv` stay pinned in the `Dockerfile`; bump them in a PR with
|
||||
the checksum from the release's `gh_<version>_checksums.txt` and the
|
||||
`gh` and `uv` stay pinned in the image recipe; bump them with the
|
||||
checksum from the release's `gh_<version>_checksums.txt` and the
|
||||
tarball's `.sha256` sidecar respectively.
|
||||
- **A local build on Apple silicon only proves the image assembles.**
|
||||
Under QEMU the Claude Code binary (a Bun executable) dies with
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
#!/usr/bin/env bash
|
||||
# Daily Claude Code compatibility-matrix populator.
|
||||
#
|
||||
# Runs daily as the Render cron job `litellm-compat-matrix`, built from
|
||||
# the Dockerfile in this directory (see README.md). The flow is:
|
||||
# Runs daily as the Render cron job `litellm-compat-matrix` (see
|
||||
# README.md for the image it runs in). The flow is:
|
||||
#
|
||||
# 1. Resolve the latest LiteLLM final release tag from the GitHub
|
||||
# Releases API.
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Image-level regression net for the prisma bake in the componentized images.
|
||||
"""Image-level regression net for the prisma bake in the `gateway` and `backend` components.
|
||||
|
||||
The gateway and backend serve requests; they never shell out to the Prisma CLI
|
||||
(``PrismaManager.setup_database`` is reachable only from ``proxy_cli.py``, which
|
||||
|
|
@ -35,7 +35,8 @@ import pytest
|
|||
IMAGE = os.getenv("LITELLM_IMAGE")
|
||||
POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine")
|
||||
CURL_IMAGE = os.getenv("LITELLM_TEST_CURL_IMAGE", "curlimages/curl:8.11.1")
|
||||
COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", "4000")
|
||||
COMPONENT = os.getenv("LITELLM_COMPONENT", "gateway")
|
||||
COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", {"gateway": "4000", "backend": "4001"}[COMPONENT])
|
||||
NON_ROOT_UID = "12345:0"
|
||||
STARTUP_TIMEOUT_SECONDS = int(os.getenv("LITELLM_COMPONENT_STARTUP_TIMEOUT", "180"))
|
||||
|
||||
|
|
@ -63,7 +64,7 @@ def offline_stack():
|
|||
|
||||
The container runs with DISABLE_SCHEMA_UPDATE, since applying the schema is
|
||||
the migration job's responsibility in this topology and needs the Prisma CLI
|
||||
these images deliberately omit, and with LITELLM_LOCAL_MODEL_COST_MAP, or the
|
||||
these components never invoke, and with LITELLM_LOCAL_MODEL_COST_MAP, or the
|
||||
proxy spends the whole startup budget timing out on a cost-map fetch over the
|
||||
network it does not have.
|
||||
|
||||
|
|
@ -92,7 +93,7 @@ def offline_stack():
|
|||
"-e", "LITELLM_MASTER_KEY=sk-component-serve-test",
|
||||
"-e", "DISABLE_SCHEMA_UPDATE=true",
|
||||
"-e", "LITELLM_LOCAL_MODEL_COST_MAP=True",
|
||||
IMAGE,
|
||||
IMAGE, COMPONENT,
|
||||
)
|
||||
yield network, component
|
||||
finally:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
"""Image-level regression net for the prisma bake in the shipped runtime image.
|
||||
|
||||
Boots a built image's migration entrypoint the way an OpenShift / air-gapped
|
||||
Boots a built image's `migrations` component the way an OpenShift / air-gapped
|
||||
deployment does (an internal-only network with no egress, an arbitrary non-root
|
||||
uid in GID 0) against a brand-new Postgres, and asserts the schema was created.
|
||||
|
||||
|
|
@ -14,6 +14,7 @@ the normal unit-test run and exercised only where an image has been built (the
|
|||
image-scan workflow). Requires a working docker CLI.
|
||||
"""
|
||||
|
||||
import shlex
|
||||
import shutil
|
||||
import subprocess
|
||||
import uuid
|
||||
|
|
@ -26,10 +27,7 @@ POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine")
|
|||
MIN_TABLES = int(os.getenv("LITELLM_TEST_MIN_TABLES", "20"))
|
||||
NON_ROOT_UID = "12345:0" # arbitrary uid in GID 0, as OpenShift restricted-v2 assigns
|
||||
|
||||
MIGRATION_INTERPRETER = os.getenv("LITELLM_MIGRATION_INTERPRETER", "python")
|
||||
MIGRATION_SCRIPT = os.getenv(
|
||||
"LITELLM_MIGRATION_SCRIPT", "litellm/proxy/prisma_migration.py"
|
||||
)
|
||||
MIGRATION_ARGS = tuple(shlex.split(os.getenv("LITELLM_MIGRATION_ARGS", "migrations")))
|
||||
|
||||
pytestmark = [
|
||||
pytest.mark.skipif(IMAGE is None, reason="requires a built image (set LITELLM_IMAGE)"),
|
||||
|
|
@ -113,8 +111,7 @@ def test_migration_offline_as_non_root_uid(offline_postgres):
|
|||
"-e", f"DATABASE_URL=postgresql://postgres:pw@{pg}:5432/litellm",
|
||||
"-e", "LITELLM_MASTER_KEY=sk-offline-migration-test",
|
||||
"-e", "DISABLE_SCHEMA_UPDATE=false",
|
||||
"-w", "/app", "--entrypoint", MIGRATION_INTERPRETER,
|
||||
IMAGE, MIGRATION_SCRIPT,
|
||||
IMAGE, *MIGRATION_ARGS,
|
||||
check=False,
|
||||
)
|
||||
tables = _table_count(pg)
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Image-level regression net for arbitrary-uid boot of the UI image.
|
||||
"""Image-level regression net for arbitrary-uid boot of the `ui` component.
|
||||
|
||||
OpenShift ``restricted-v2`` ignores the image ``USER`` and assigns an
|
||||
arbitrary uid in GID 0. The stock nginx base expects to start as root, so
|
||||
arbitrary uid in GID 0. A stock nginx install expects to start as root, so
|
||||
its cache (``/var/cache/nginx``) and pid (``/run``) paths are root-owned
|
||||
755 and the master process dies at startup with
|
||||
``mkdir() "/var/cache/nginx/client_temp" failed (13: Permission denied)``.
|
||||
|
|
@ -63,7 +63,7 @@ def ui_container() -> Iterator[tuple[str, str]]:
|
|||
"run", "-d", "--name", container, "--network", network,
|
||||
"--user", ARBITRARY_UID,
|
||||
"--read-only", "--tmpfs", "/tmp",
|
||||
IMAGE,
|
||||
IMAGE, "ui",
|
||||
)
|
||||
yield network, container
|
||||
finally:
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
"""Unit tests for `docker/component_entrypoint.sh` and its wiring into the
|
||||
componentized `gateway` / `backend` images and Terraform deployments."""
|
||||
"""Unit tests for `docker-entrypoint.sh`, the component dispatcher every shipped
|
||||
container runs, and for the Terraform launch commands that must agree with it."""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
|
@ -11,15 +11,13 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
COMPONENT_ENTRYPOINT = REPO_ROOT / "docker" / "component_entrypoint.sh"
|
||||
PROD_ENTRYPOINT = REPO_ROOT / "docker" / "prod_entrypoint.sh"
|
||||
GATEWAY_DOCKERFILE = REPO_ROOT / "gateway" / "Dockerfile"
|
||||
BACKEND_DOCKERFILE = REPO_ROOT / "backend" / "Dockerfile"
|
||||
BUILD_FROM_PIP_DOCKERFILE = REPO_ROOT / "docker" / "build_from_pip" / "Dockerfile.build_from_pip"
|
||||
DOCKER_ENTRYPOINT = REPO_ROOT / "docker-entrypoint.sh"
|
||||
DOCKERFILE = REPO_ROOT / "Dockerfile"
|
||||
TERRAFORM_ECS = REPO_ROOT / "terraform" / "litellm" / "aws" / "ecs.tf"
|
||||
TERRAFORM_CLOUDRUN = REPO_ROOT / "terraform" / "litellm" / "gcp" / "cloudrun.tf"
|
||||
|
||||
IMAGE_ENTRYPOINT_PATH = "/app/docker/component_entrypoint.sh"
|
||||
IMAGE_ENTRYPOINT_PATH = "/app/docker-entrypoint.sh"
|
||||
STUBBED_EXECUTABLES = ("ddtrace-run", "uvicorn", "python", "litellm", "nginx")
|
||||
|
||||
TRUTHY_USE_DDTRACE = ("true", "True", "TRUE", "tRuE")
|
||||
FALSY_USE_DDTRACE = (None, "", "false", "False", "1", "yes", "on", "truex")
|
||||
|
|
@ -37,7 +35,6 @@ _STUB_TEMPLATE = """#!/bin/sh
|
|||
|
||||
_ENTRYPOINT_RE = re.compile(r"^ENTRYPOINT\s+(\[.*\])\s*$", re.MULTILINE)
|
||||
_CMD_RE = re.compile(r"^CMD\s+(\[.*\])\s*$", re.MULTILINE)
|
||||
_COPY_RE = re.compile(r"^COPY\s+(?!--from)(\S+)\s+(\S+)\s*$", re.MULTILINE)
|
||||
_APP_TARGET_RE = re.compile(r"(?:gateway|backend)\.main:app|gateway\.launch")
|
||||
_TF_STRING_LOCAL_RE = re.compile(r'^\s*(\w+)\s*=\s*"((?:[^"\\]|\\.)*)"\s*$', re.MULTILINE)
|
||||
_TF_INTERPOLATION_RE = re.compile(r"\$\{(local|var)\.(\w+)\}")
|
||||
|
|
@ -58,53 +55,39 @@ def _write_stubs(bin_dir: Path, names: tuple[str, ...]) -> None:
|
|||
stub.chmod(0o755)
|
||||
|
||||
|
||||
def _run_entrypoint(
|
||||
script: Path,
|
||||
argv: tuple[str, ...],
|
||||
use_ddtrace: str | None,
|
||||
tmp_path: Path,
|
||||
) -> tuple[str, ...]:
|
||||
"""Run `script` with stubbed executables on PATH and return the recorded lines."""
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir(parents=True)
|
||||
_write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python", "litellm"))
|
||||
record = tmp_path / "record.txt"
|
||||
|
||||
env = {
|
||||
**os.environ,
|
||||
def _entrypoint_env(bin_dir: Path, record: Path, overrides: dict[str, str | None]) -> dict[str, str]:
|
||||
"""The container-like environment the entrypoint runs under, with `overrides` applied (None unsets)."""
|
||||
cleared = ("USE_DDTRACE", "DD_TRACE_OPENAI_ENABLED", "LITELLM_COMPONENT", "NUM_WORKERS", *overrides)
|
||||
return {
|
||||
**{k: v for k, v in os.environ.items() if k not in cleared},
|
||||
**{k: v for k, v in overrides.items() if v is not None},
|
||||
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
|
||||
"RECORD": str(record),
|
||||
"PYTHONPATH": PYTHONPATH_SENTINEL,
|
||||
}
|
||||
env.pop("USE_DDTRACE", None)
|
||||
env.pop("DD_TRACE_OPENAI_ENABLED", None)
|
||||
if use_ddtrace is not None:
|
||||
env["USE_DDTRACE"] = use_ddtrace
|
||||
|
||||
result = subprocess.run(
|
||||
["sh", str(script), *argv],
|
||||
env=env,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
|
||||
def _run_entrypoint(
|
||||
argv: tuple[str, ...],
|
||||
tmp_path: Path,
|
||||
use_ddtrace: str | None = None,
|
||||
**overrides: str | None,
|
||||
) -> tuple[str, ...]:
|
||||
"""Run the entrypoint with stubbed executables on PATH and return the recorded lines."""
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir(parents=True)
|
||||
_write_stubs(bin_dir, STUBBED_EXECUTABLES)
|
||||
record = tmp_path / "record.txt"
|
||||
env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace, **overrides})
|
||||
|
||||
result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT), *argv], env=env, capture_output=True, text=True, check=False)
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
return tuple(record.read_text().splitlines()) if record.exists() else ()
|
||||
|
||||
|
||||
def _run_shell_command(command: str, bin_dir: Path, record: Path, use_ddtrace: str | None) -> tuple[str, ...]:
|
||||
"""Run a resolved Terraform launch command through `sh -c` and return the recorded lines."""
|
||||
env = {
|
||||
**os.environ,
|
||||
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
|
||||
"RECORD": str(record),
|
||||
"PYTHONPATH": PYTHONPATH_SENTINEL,
|
||||
}
|
||||
env.pop("USE_DDTRACE", None)
|
||||
env.pop("DD_TRACE_OPENAI_ENABLED", None)
|
||||
if use_ddtrace is not None:
|
||||
env["USE_DDTRACE"] = use_ddtrace
|
||||
|
||||
env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace})
|
||||
result = subprocess.run(["sh", "-c", command], env=env, capture_output=True, text=True, check=False)
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
return tuple(record.read_text().splitlines()) if record.exists() else ()
|
||||
|
|
@ -131,29 +114,137 @@ def _resolve_tf_local(terraform_file: Path, name: str) -> str:
|
|||
def _entrypoint_argv(dockerfile: Path) -> tuple[str, ...]:
|
||||
matches = _ENTRYPOINT_RE.findall(dockerfile.read_text())
|
||||
assert matches, f"no exec-form ENTRYPOINT found in {dockerfile}"
|
||||
parsed = json.loads(matches[-1])
|
||||
return tuple(str(part) for part in parsed)
|
||||
|
||||
|
||||
def _cmd_argv(dockerfile: Path) -> tuple[str, ...]:
|
||||
matches = _CMD_RE.findall(dockerfile.read_text())
|
||||
assert matches, f"no exec-form CMD found in {dockerfile}"
|
||||
return tuple(str(part) for part in json.loads(matches[-1]))
|
||||
|
||||
|
||||
def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(
|
||||
tmp_path: Path,
|
||||
@pytest.mark.parametrize(
|
||||
"argv, expected_exec, expected_args",
|
||||
[
|
||||
(("proxy", "--config", "/app/config.yaml"), "exec=litellm", "args=--config /app/config.yaml"),
|
||||
(("gateway",), "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"),
|
||||
(
|
||||
("gateway", "--port", "8080"),
|
||||
"exec=python",
|
||||
"args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 8080",
|
||||
),
|
||||
(("backend",), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"),
|
||||
(("backend", "--port", "9001"), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001 --port 9001"),
|
||||
(("ui",), "exec=nginx", "args=-g daemon off;"),
|
||||
(("migrations",), "exec=python", "args=/app/migrations/run.py"),
|
||||
(("metrics", "--port", "9090"), "exec=python", "args=-m litellm.proxy.prometheus_metrics_server --port 9090"),
|
||||
(("collector",), "exec=python", "args=-m litellm.proxy.collector"),
|
||||
],
|
||||
)
|
||||
def test_the_first_argument_selects_the_component(
|
||||
argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path
|
||||
) -> None:
|
||||
recorded = _run_entrypoint(argv, tmp_path)
|
||||
|
||||
assert recorded[:2] == (expected_exec, expected_args)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"argv, expected",
|
||||
[
|
||||
((), ("exec=litellm", "args=")),
|
||||
(("--port", "4000"), ("exec=litellm", "args=--port 4000")),
|
||||
(
|
||||
("--config", "/app/config.yaml", "--detailed_debug"),
|
||||
("exec=litellm", "args=--config /app/config.yaml --detailed_debug"),
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_flags_alone_still_run_the_monolithic_proxy(
|
||||
argv: tuple[str, ...], expected: tuple[str, str], tmp_path: Path
|
||||
) -> None:
|
||||
"""Every `docker run litellm --config ...` written before components existed keeps working."""
|
||||
assert _run_entrypoint(argv, tmp_path)[:2] == expected
|
||||
|
||||
|
||||
def test_the_dockerfile_leaves_the_command_empty_so_the_env_var_can_pick_the_component(tmp_path: Path) -> None:
|
||||
"""A CMD naming a component would beat LITELLM_COMPONENT, and the proxy already listens on 4000 with no flags."""
|
||||
dockerfile_text = DOCKERFILE.read_text()
|
||||
assert _entrypoint_argv(DOCKERFILE) == (IMAGE_ENTRYPOINT_PATH,), "the image ENTRYPOINT must be the bare dispatcher"
|
||||
assert not _CMD_RE.search(dockerfile_text), "the Dockerfile must not set a CMD"
|
||||
|
||||
assert _run_entrypoint((), tmp_path)[:2] == ("exec=litellm", "args=")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"component, argv, expected_exec, expected_args",
|
||||
[
|
||||
("backend", (), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"),
|
||||
(
|
||||
"gateway",
|
||||
("--port", "4100"),
|
||||
"exec=python",
|
||||
"args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 4100",
|
||||
),
|
||||
("ui", (), "exec=nginx", "args=-g daemon off;"),
|
||||
],
|
||||
)
|
||||
def test_litellm_component_env_selects_the_component_when_the_command_has_none(
|
||||
component: str, argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path
|
||||
) -> None:
|
||||
"""Helm and Compose can pick the component with an env var and keep `args` for the process flags."""
|
||||
recorded = _run_entrypoint(argv, tmp_path, LITELLM_COMPONENT=component)
|
||||
|
||||
assert recorded[:2] == (expected_exec, expected_args)
|
||||
|
||||
|
||||
def test_an_explicit_component_argument_beats_the_env_var(tmp_path: Path) -> None:
|
||||
recorded = _run_entrypoint(("backend",), tmp_path, LITELLM_COMPONENT="gateway")
|
||||
|
||||
assert recorded[0] == "exec=uvicorn"
|
||||
|
||||
|
||||
def test_the_gateway_honours_num_workers(tmp_path: Path) -> None:
|
||||
recorded = _run_entrypoint(("gateway",), tmp_path, NUM_WORKERS="4")
|
||||
|
||||
assert recorded[1] == "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"
|
||||
|
||||
|
||||
def test_an_unknown_first_word_is_run_as_the_command(tmp_path: Path) -> None:
|
||||
"""`docker run <image> sh -c ...` and `kubectl exec`-style overrides bypass the dispatcher."""
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir()
|
||||
_write_stubs(bin_dir, ("some-tool",))
|
||||
record = tmp_path / "record.txt"
|
||||
env = _entrypoint_env(bin_dir, record, {})
|
||||
|
||||
result = subprocess.run(
|
||||
["sh", str(DOCKER_ENTRYPOINT), "some-tool", "--flag"], env=env, capture_output=True, text=True, check=False
|
||||
)
|
||||
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
assert record.read_text().splitlines()[:2] == ["exec=some-tool", "args=--flag"]
|
||||
|
||||
|
||||
def test_an_unknown_litellm_component_fails_fast_instead_of_guessing(tmp_path: Path) -> None:
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir()
|
||||
_write_stubs(bin_dir, STUBBED_EXECUTABLES)
|
||||
record = tmp_path / "record.txt"
|
||||
env = _entrypoint_env(bin_dir, record, {"LITELLM_COMPONENT": "gatway"})
|
||||
|
||||
result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT)], env=env, capture_output=True, text=True, check=False)
|
||||
|
||||
assert result.returncode == 64
|
||||
assert "gatway" in result.stderr
|
||||
assert not record.exists(), "nothing may be exec'd when the component is unknown"
|
||||
|
||||
|
||||
def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(tmp_path: Path) -> None:
|
||||
"""`USE_DDTRACE=true` must prefix the command with `ddtrace-run`, turn the openai
|
||||
integration off, and leave PYTHONPATH alone.
|
||||
|
||||
`ddtrace-run` installs its instrumentation by PREPENDING a bootstrap directory to
|
||||
PYTHONPATH, and the images set PYTHONPATH=/app so the app package is importable. A
|
||||
wrapper that assigned PYTHONPATH instead of inheriting it would either drop the
|
||||
bootstrap (silently disabling tracing) or drop /app (breaking the import), so the
|
||||
recorded value is asserted verbatim.
|
||||
PYTHONPATH, and the image sets PYTHONPATH=/app so the component packages are
|
||||
importable. A wrapper that assigned PYTHONPATH instead of inheriting it would either
|
||||
drop the bootstrap (silently disabling tracing) or drop /app (breaking the import), so
|
||||
the recorded value is asserted verbatim.
|
||||
|
||||
PYTHONPATH_SENTINEL deliberately differs from the images' own /app: with /app as the
|
||||
PYTHONPATH_SENTINEL deliberately differs from the image's own /app: with /app as the
|
||||
fixture value, a wrapper that overwrote PYTHONPATH with /app would still satisfy this
|
||||
assertion and the check would prove nothing.
|
||||
|
||||
|
|
@ -161,32 +252,22 @@ def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(
|
|||
it before any litellm code runs, so litellm's in-process `patch_all(..., openai=False)`
|
||||
can no longer suppress it; leaving it on double-reports every LLM call.
|
||||
"""
|
||||
recorded = _run_entrypoint(
|
||||
COMPONENT_ENTRYPOINT,
|
||||
("uvicorn", "gateway.main:app", "--workers", "2", "--port", "4000"),
|
||||
use_ddtrace="true",
|
||||
tmp_path=tmp_path,
|
||||
)
|
||||
recorded = _run_entrypoint(("gateway",), tmp_path, use_ddtrace="true", NUM_WORKERS="2")
|
||||
|
||||
assert recorded == (
|
||||
"exec=ddtrace-run",
|
||||
"args=uvicorn gateway.main:app --workers 2 --port 4000",
|
||||
"args=python -m gateway.launch --workers 2 --host 0.0.0.0 --port 4000",
|
||||
"DD_TRACE_OPENAI_ENABLED=False",
|
||||
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
|
||||
)
|
||||
|
||||
|
||||
def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None:
|
||||
recorded = _run_entrypoint(
|
||||
COMPONENT_ENTRYPOINT,
|
||||
("uvicorn", "backend.main:app", "--port", "4001"),
|
||||
use_ddtrace=None,
|
||||
tmp_path=tmp_path,
|
||||
)
|
||||
recorded = _run_entrypoint(("backend", "--workers", "2"), tmp_path)
|
||||
|
||||
assert recorded == (
|
||||
"exec=uvicorn",
|
||||
"args=backend.main:app --port 4001",
|
||||
"args=backend.main:app --host 0.0.0.0 --port 4001 --workers 2",
|
||||
"DD_TRACE_OPENAI_ENABLED=<unset>",
|
||||
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
|
||||
)
|
||||
|
|
@ -196,35 +277,24 @@ def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None:
|
|||
"use_ddtrace, traced",
|
||||
[*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)],
|
||||
)
|
||||
def test_gating_matches_the_monolithic_entrypoint_and_get_secret_bool(
|
||||
def test_ddtrace_gating_matches_get_secret_bool_for_every_component(
|
||||
use_ddtrace: str | None, traced: bool, tmp_path: Path
|
||||
) -> None:
|
||||
"""Both entrypoints must accept exactly the spellings `get_secret_bool` accepts.
|
||||
"""The shell gate must accept exactly the spellings `get_secret_bool` accepts.
|
||||
|
||||
`ProxyStartupEvent._init_dd_tracer` reads `USE_DDTRACE` through `get_secret_bool`, which
|
||||
matches `true` case-insensitively. If the shell gate were stricter, `USE_DDTRACE=True` would
|
||||
give in-process LLM spans without `ddtrace-run` HTTP spans, a half-enabled state.
|
||||
"""
|
||||
component = _run_entrypoint(
|
||||
COMPONENT_ENTRYPOINT,
|
||||
("uvicorn", "gateway.main:app"),
|
||||
use_ddtrace=use_ddtrace,
|
||||
tmp_path=tmp_path / "component",
|
||||
)
|
||||
monolith = _run_entrypoint(
|
||||
PROD_ENTRYPOINT,
|
||||
("--port", "4000"),
|
||||
use_ddtrace=use_ddtrace,
|
||||
tmp_path=tmp_path / "monolith",
|
||||
)
|
||||
component = _run_entrypoint(("gateway",), tmp_path / "component", use_ddtrace=use_ddtrace)
|
||||
monolith = _run_entrypoint(("--port", "4000"), tmp_path / "monolith", use_ddtrace=use_ddtrace)
|
||||
|
||||
expected_exec = "exec=ddtrace-run" if traced else "exec=uvicorn"
|
||||
expected_openai = "DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=<unset>"
|
||||
assert component[0] == expected_exec
|
||||
assert component[0] == ("exec=ddtrace-run" if traced else "exec=python")
|
||||
assert component[2] == expected_openai
|
||||
assert monolith[0] == ("exec=ddtrace-run" if traced else "exec=litellm")
|
||||
assert monolith[2] == expected_openai
|
||||
assert monolith[1] == ("args=litellm --port 4000" if traced else "args=--port 4000")
|
||||
assert monolith[2] == expected_openai
|
||||
|
||||
|
||||
def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path) -> None:
|
||||
|
|
@ -236,19 +306,26 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path)
|
|||
(multiproc_dir / "counter_7.db").write_bytes(b"stale")
|
||||
(multiproc_dir / "keep.txt").write_text("not a sample")
|
||||
|
||||
recorded = _run_entrypoint(("gateway",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir))
|
||||
|
||||
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"]
|
||||
assert recorded[0] == "exec=python"
|
||||
|
||||
|
||||
def test_a_raw_command_leaves_the_prometheus_multiproc_dir_untouched(tmp_path: Path) -> None:
|
||||
"""The entrypoint cannot tell a raw writer from a raw reader, so `docker run <image> python -m
|
||||
litellm.proxy.prometheus_metrics_server` must not delete the samples the gateway workers are still writing."""
|
||||
multiproc_dir = tmp_path / "multiproc"
|
||||
multiproc_dir.mkdir()
|
||||
(multiproc_dir / "counter_7.db").write_bytes(b"live")
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir()
|
||||
_write_stubs(bin_dir, ("uvicorn",))
|
||||
_write_stubs(bin_dir, ("python",))
|
||||
record = tmp_path / "record.txt"
|
||||
env = {
|
||||
**os.environ,
|
||||
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
|
||||
"RECORD": str(record),
|
||||
"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir),
|
||||
}
|
||||
env.pop("USE_DDTRACE", None)
|
||||
env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)})
|
||||
|
||||
result = subprocess.run(
|
||||
["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"],
|
||||
["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "litellm.proxy.prometheus_metrics_server"],
|
||||
env=env,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
|
|
@ -256,161 +333,69 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path)
|
|||
)
|
||||
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"]
|
||||
assert record.read_text().splitlines()[0] == "exec=uvicorn"
|
||||
assert [p.name for p in multiproc_dir.iterdir()] == ["counter_7.db"]
|
||||
assert record.read_text().splitlines()[0] == "exec=python"
|
||||
|
||||
|
||||
def test_a_raw_command_still_gets_the_prometheus_multiproc_dir_created(tmp_path: Path) -> None:
|
||||
"""`docker run <image> python -m gateway.launch` cannot write samples into a directory that is not there."""
|
||||
multiproc_dir = tmp_path / "not-yet" / "multiproc"
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir()
|
||||
_write_stubs(bin_dir, ("python",))
|
||||
record = tmp_path / "record.txt"
|
||||
env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)})
|
||||
|
||||
result = subprocess.run(
|
||||
["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "gateway.launch"],
|
||||
env=env,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
assert multiproc_dir.is_dir()
|
||||
assert record.read_text().splitlines()[0] == "exec=python"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("component", ["metrics", "collector"])
|
||||
def test_readers_of_the_prometheus_multiproc_dir_keep_the_workers_samples(component: str, tmp_path: Path) -> None:
|
||||
"""The metrics and collector sidecars share the volume with gateway workers that are already
|
||||
serving traffic, so their (re)start must not erase the samples those workers have written."""
|
||||
multiproc_dir = tmp_path / "multiproc"
|
||||
multiproc_dir.mkdir()
|
||||
(multiproc_dir / "counter_12.db").write_bytes(b"live")
|
||||
(multiproc_dir / "histogram_13.db").write_bytes(b"live")
|
||||
|
||||
recorded = _run_entrypoint((component,), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir))
|
||||
|
||||
assert sorted(p.name for p in multiproc_dir.iterdir()) == ["counter_12.db", "histogram_13.db"]
|
||||
assert recorded[0] == "exec=python"
|
||||
|
||||
|
||||
def test_creates_a_missing_prometheus_multiproc_dir(tmp_path: Path) -> None:
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir()
|
||||
_write_stubs(bin_dir, ("uvicorn",))
|
||||
missing = tmp_path / "multiproc"
|
||||
env = {
|
||||
**os.environ,
|
||||
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
|
||||
"RECORD": str(tmp_path / "record.txt"),
|
||||
"PROMETHEUS_MULTIPROC_DIR": str(missing),
|
||||
}
|
||||
env.pop("USE_DDTRACE", None)
|
||||
result = subprocess.run(
|
||||
["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"], env=env, capture_output=True, text=True
|
||||
)
|
||||
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
_run_entrypoint(("backend",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(missing))
|
||||
|
||||
assert missing.is_dir()
|
||||
|
||||
|
||||
def _copied_script(dockerfile: Path, image_path: str) -> Path:
|
||||
"""Resolve the repo file a Dockerfile `COPY`s to `image_path`, so tests run what the image ships."""
|
||||
matches = _COPY_RE.findall(dockerfile.read_text())
|
||||
sources = tuple(src for src, dst in matches if dst == image_path)
|
||||
assert sources, f"{dockerfile} never COPYs anything to {image_path}"
|
||||
source = REPO_ROOT / sources[-1]
|
||||
assert source.is_file(), f"{dockerfile} COPYs {sources[-1]}, which does not exist in the build context"
|
||||
return source
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"use_ddtrace, traced",
|
||||
[*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)],
|
||||
)
|
||||
def test_build_from_pip_image_launches_litellm_through_the_prod_entrypoint(
|
||||
use_ddtrace: str | None, traced: bool, tmp_path: Path
|
||||
) -> None:
|
||||
"""Run the build_from_pip image's ENTRYPOINT + CMD through the script it actually COPYs.
|
||||
|
||||
The image used to `ENTRYPOINT ["litellm"]`, so `USE_DDTRACE` was inert there at any spelling.
|
||||
Resolving the ENTRYPOINT path back to its COPY source and executing it with the Dockerfile's
|
||||
CMD checks the launch the container performs, not just that the Dockerfile mentions the script.
|
||||
"""
|
||||
entrypoint = _entrypoint_argv(BUILD_FROM_PIP_DOCKERFILE)
|
||||
assert len(entrypoint) == 1, (
|
||||
f"{BUILD_FROM_PIP_DOCKERFILE} ENTRYPOINT must be the bare script so CMD reaches litellm"
|
||||
)
|
||||
script = _copied_script(BUILD_FROM_PIP_DOCKERFILE, entrypoint[0])
|
||||
assert script == PROD_ENTRYPOINT, f"{BUILD_FROM_PIP_DOCKERFILE} bypasses the ddtrace-aware entrypoint"
|
||||
assert f"chmod +x {entrypoint[0]}" in BUILD_FROM_PIP_DOCKERFILE.read_text()
|
||||
|
||||
cmd = _cmd_argv(BUILD_FROM_PIP_DOCKERFILE)
|
||||
recorded = _run_entrypoint(script, cmd, use_ddtrace=use_ddtrace, tmp_path=tmp_path)
|
||||
|
||||
cmd_str = " ".join(cmd)
|
||||
assert recorded == (
|
||||
"exec=ddtrace-run" if traced else "exec=litellm",
|
||||
f"args=litellm {cmd_str}" if traced else f"args={cmd_str}",
|
||||
"DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=<unset>",
|
||||
f"PYTHONPATH={PYTHONPATH_SENTINEL}",
|
||||
)
|
||||
|
||||
|
||||
def test_entrypoint_script_is_executable() -> None:
|
||||
mode = COMPONENT_ENTRYPOINT.stat().st_mode
|
||||
assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT"
|
||||
assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user"
|
||||
|
||||
|
||||
def test_entrypoint_script_has_no_carriage_returns() -> None:
|
||||
assert b"\r" not in COMPONENT_ENTRYPOINT.read_bytes()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"dockerfile, launcher",
|
||||
[
|
||||
(GATEWAY_DOCKERFILE, "python -m gateway.launch"),
|
||||
(BACKEND_DOCKERFILE, "uvicorn backend.main:app"),
|
||||
],
|
||||
)
|
||||
def test_component_images_launch_uvicorn_through_the_entrypoint(dockerfile: Path, launcher: str) -> None:
|
||||
entrypoint = " ".join(_entrypoint_argv(dockerfile))
|
||||
|
||||
assert IMAGE_ENTRYPOINT_PATH in entrypoint, f"{dockerfile} bypasses the ddtrace-aware entrypoint"
|
||||
assert launcher in entrypoint
|
||||
assert entrypoint.index(IMAGE_ENTRYPOINT_PATH) < entrypoint.index(launcher), (
|
||||
f"{dockerfile} must invoke uvicorn through the entrypoint, not the other way around"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"use_ddtrace, num_workers, expected_exec, expected_args",
|
||||
[
|
||||
(None, "4", "exec=python", "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"),
|
||||
(None, None, "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"),
|
||||
("true", "4", "exec=ddtrace-run", "args=python -m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"),
|
||||
],
|
||||
)
|
||||
def test_gateway_image_execs_the_supervisor_with_its_worker_count(
|
||||
use_ddtrace: str | None, num_workers: str | None, expected_exec: str, expected_args: str, tmp_path: Path
|
||||
) -> None:
|
||||
"""Run the gateway image's ENTRYPOINT + CMD and record what the container execs.
|
||||
|
||||
The Dockerfile's `/app/...` script path is resolved to the checked-in script and `python`
|
||||
is stubbed on PATH, so the assertion is on the argv `gateway.launch` receives, not on the
|
||||
Dockerfile text.
|
||||
"""
|
||||
entrypoint = tuple(
|
||||
part.replace(IMAGE_ENTRYPOINT_PATH, str(COMPONENT_ENTRYPOINT)) for part in _entrypoint_argv(GATEWAY_DOCKERFILE)
|
||||
)
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir(parents=True)
|
||||
_write_stubs(bin_dir, ("ddtrace-run", "python", "uvicorn"))
|
||||
record = tmp_path / "record.txt"
|
||||
overrides = {"USE_DDTRACE": use_ddtrace, "NUM_WORKERS": num_workers}
|
||||
env = {
|
||||
**{k: v for k, v in os.environ.items() if k not in ("DD_TRACE_OPENAI_ENABLED", *overrides)},
|
||||
**{k: v for k, v in overrides.items() if v is not None},
|
||||
"PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}",
|
||||
"RECORD": str(record),
|
||||
"PYTHONPATH": PYTHONPATH_SENTINEL,
|
||||
}
|
||||
|
||||
result = subprocess.run(
|
||||
[*entrypoint, *_cmd_argv(GATEWAY_DOCKERFILE)], env=env, capture_output=True, text=True, check=False
|
||||
)
|
||||
|
||||
assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}"
|
||||
assert tuple(record.read_text().splitlines())[:2] == (expected_exec, expected_args)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("dockerfile", [GATEWAY_DOCKERFILE, BACKEND_DOCKERFILE])
|
||||
def test_component_images_make_the_entrypoint_executable(dockerfile: Path) -> None:
|
||||
body = dockerfile.read_text()
|
||||
assert "chmod +x docker/component_entrypoint.sh" in body
|
||||
|
||||
|
||||
@pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name)
|
||||
@pytest.mark.parametrize("component", ["gateway", "backend"])
|
||||
@pytest.mark.parametrize("use_ddtrace", [*TRUTHY_USE_DDTRACE, *FALSY_USE_DDTRACE])
|
||||
def test_terraform_launch_command_matches_the_script_contract(
|
||||
terraform_file: Path, component: str, use_ddtrace: str | None, tmp_path: Path
|
||||
) -> None:
|
||||
"""The Terraform command and `docker/component_entrypoint.sh` must decide identically.
|
||||
"""The Terraform command and `docker-entrypoint.sh <component>` must decide identically.
|
||||
|
||||
The decision deliberately lives in two places. The script is what the image ENTRYPOINT runs;
|
||||
the Terraform strings are what runs when a deployment overrides that ENTRYPOINT, and they
|
||||
cannot call the script because the caller supplies the image tag and it may predate the file.
|
||||
Both modules default to a tag that does. So instead of asserting a shared path, this runs both
|
||||
implementations under the same environment and asserts they agree on which binary is exec'd
|
||||
and on whether the openai integration is disabled.
|
||||
So instead of asserting a shared path, this runs both implementations under the same
|
||||
environment and asserts they agree on which binary is exec'd and on whether the openai
|
||||
integration is disabled.
|
||||
"""
|
||||
launcher = COMPONENT_LAUNCHERS[component]
|
||||
app_target = " ".join(launcher[1:])
|
||||
|
|
@ -421,18 +406,14 @@ def test_terraform_launch_command_matches_the_script_contract(
|
|||
_write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python"))
|
||||
from_terraform = _run_shell_command(command, bin_dir, tmp_path / "terraform.txt", use_ddtrace)
|
||||
|
||||
from_script = _run_entrypoint(
|
||||
COMPONENT_ENTRYPOINT,
|
||||
launcher,
|
||||
use_ddtrace=use_ddtrace,
|
||||
tmp_path=tmp_path / "script",
|
||||
)
|
||||
from_script = _run_entrypoint((component,), tmp_path / "script", use_ddtrace=use_ddtrace)
|
||||
|
||||
assert from_terraform[0] == from_script[0], (
|
||||
f"{terraform_file} disagrees with the script on USE_DDTRACE={use_ddtrace}"
|
||||
)
|
||||
assert from_terraform[2] == from_script[2], f"{terraform_file} disagrees with the script on the openai integration"
|
||||
assert app_target in from_terraform[1]
|
||||
assert app_target in from_script[1]
|
||||
assert "gateway.main:app" not in from_terraform[1], f"{terraform_file} bypasses the gateway.launch supervisor"
|
||||
|
||||
if use_ddtrace in TRUTHY_USE_DDTRACE:
|
||||
|
|
@ -478,9 +459,11 @@ def test_terraform_does_not_depend_on_the_entrypoint_script(terraform_file: Path
|
|||
)
|
||||
|
||||
|
||||
def test_gateway_keeps_its_worker_count_and_backend_keeps_a_single_process() -> None:
|
||||
gateway = " ".join(_entrypoint_argv(GATEWAY_DOCKERFILE)) + " " + " ".join(_cmd_argv(GATEWAY_DOCKERFILE))
|
||||
backend = " ".join(_entrypoint_argv(BACKEND_DOCKERFILE)) + " " + " ".join(_cmd_argv(BACKEND_DOCKERFILE))
|
||||
def test_entrypoint_script_is_executable() -> None:
|
||||
mode = DOCKER_ENTRYPOINT.stat().st_mode
|
||||
assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT"
|
||||
assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user"
|
||||
|
||||
assert "--workers" in gateway and "NUM_WORKERS" in gateway
|
||||
assert "--workers" not in backend and "NUM_WORKERS" not in backend
|
||||
|
||||
def test_entrypoint_script_has_no_carriage_returns() -> None:
|
||||
assert b"\r" not in DOCKER_ENTRYPOINT.read_bytes()
|
||||
|
|
@ -1,5 +1,5 @@
|
|||
"""
|
||||
Static checks that every proxy Docker image installs the `bedrock-realtime` extra.
|
||||
Static checks that the shipped Docker image installs the `bedrock-realtime` extra.
|
||||
|
||||
Bedrock Nova Sonic speech-to-speech (`/v1/realtime`) needs `aws-sdk-bedrock-runtime`,
|
||||
which only ships in the `bedrock-realtime` extra. An image whose `uv sync` stages
|
||||
|
|
@ -23,12 +23,7 @@ else:
|
|||
|
||||
REPO_ROOT: Final = os.path.join(os.path.dirname(__file__), "..", "..")
|
||||
|
||||
PROXY_DOCKERFILES: Final = (
|
||||
"Dockerfile",
|
||||
os.path.join("docker", "Dockerfile.non_root"),
|
||||
os.path.join("docker", "Dockerfile.database"),
|
||||
os.path.join("gateway", "Dockerfile"),
|
||||
)
|
||||
PROXY_DOCKERFILES: Final = ("Dockerfile",)
|
||||
|
||||
CONTINUED_LINE_RE: Final = re.compile(r"(?:\\\n|[^\n])+")
|
||||
UV_SYNC_BOUNDARY_RE: Final = re.compile(r"(?=uv sync)")
|
||||
|
|
|
|||
|
|
@ -1,39 +1,26 @@
|
|||
"""
|
||||
Static checks on docker/Dockerfile.non_root.
|
||||
Static checks on the shipped Dockerfile's runtime user.
|
||||
|
||||
The non_root image is intended for deployment into hardened Kubernetes
|
||||
clusters where `securityContext.runAsNonRoot: true` is enforced. The
|
||||
kubelet validates non-root status by parsing the image's USER field as
|
||||
an integer — a string name like "nobody" is rejected with
|
||||
CreateContainerConfigError because the kubelet cannot resolve
|
||||
/etc/passwd inside the image at admission time.
|
||||
The image is deployed into hardened Kubernetes clusters where
|
||||
`securityContext.runAsNonRoot: true` is enforced. The kubelet validates
|
||||
non-root status by parsing the image's USER field as an integer: a string
|
||||
name like "nonroot" is rejected with CreateContainerConfigError because the
|
||||
kubelet cannot resolve /etc/passwd inside the image at admission time.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
DOCKERFILE_PATH = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"..",
|
||||
"..",
|
||||
"docker",
|
||||
"Dockerfile.non_root",
|
||||
)
|
||||
DOCKERFILE_PATH = os.path.join(os.path.dirname(__file__), "..", "..", "Dockerfile")
|
||||
|
||||
|
||||
def _final_user_directive(dockerfile_text: str) -> str:
|
||||
"""Return the value of the last `USER` directive in the file."""
|
||||
"""Return the uid of the last `USER` directive in the file (`USER uid` or `USER uid:gid`)."""
|
||||
matches = re.findall(r"^USER\s+(\S+)\s*$", dockerfile_text, re.MULTILINE)
|
||||
assert matches, "Dockerfile.non_root has no USER directive"
|
||||
return matches[-1]
|
||||
assert matches, "Dockerfile has no USER directive"
|
||||
return matches[-1].split(":", 1)[0]
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not os.path.exists(DOCKERFILE_PATH),
|
||||
reason="Dockerfile.non_root not present in this checkout",
|
||||
)
|
||||
def test_final_user_directive_is_numeric():
|
||||
"""The runtime USER must be a numeric UID so kubelet's runAsNonRoot
|
||||
admission check (strconv.Atoi) succeeds."""
|
||||
|
|
@ -43,12 +30,9 @@ def test_final_user_directive_is_numeric():
|
|||
final_user = _final_user_directive(contents)
|
||||
|
||||
assert final_user.isdigit(), (
|
||||
f"Dockerfile.non_root final USER is {final_user!r}; must be a numeric UID "
|
||||
f"Dockerfile final USER is {final_user!r}; must be a numeric UID "
|
||||
"so Kubernetes' runAsNonRoot admission check can verify non-root status. "
|
||||
"See https://kubernetes.io/docs/tasks/configure-pod-container/security-context/"
|
||||
)
|
||||
|
||||
assert int(final_user) != 0, (
|
||||
f"Dockerfile.non_root final USER is {final_user} (root); the non_root image "
|
||||
"must run as a non-zero UID."
|
||||
)
|
||||
assert int(final_user) != 0, f"Dockerfile final USER is {final_user} (root); the image must run as a non-zero UID."
|
||||
|
|
|
|||
|
|
@ -1,42 +0,0 @@
|
|||
# syntax=docker/dockerfile:1.7
|
||||
|
||||
# UI container — Next.js static export served by nginx.
|
||||
|
||||
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
|
||||
ARG NGINX_VERSION=1.31.5-alpine3.24@sha256:34f40471dea485273c5e2a04dd5e97a682332ceb4a9adecd67de450dcb2fb390
|
||||
|
||||
# ---------- builder ----------
|
||||
FROM ${UI_BUILD_IMAGE} AS builder
|
||||
|
||||
ENV NEXT_TELEMETRY_DISABLED=1 \
|
||||
npm_config_fund=false \
|
||||
npm_config_audit=false
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Layer the lockfile-only install above the source copy so source-only
|
||||
# edits don't bust the install cache.
|
||||
COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./
|
||||
RUN --mount=type=cache,target=/root/.npm \
|
||||
npm ci --prefer-offline
|
||||
|
||||
COPY ui/litellm-dashboard/ ./
|
||||
RUN npm run build
|
||||
|
||||
# ---------- runtime ----------
|
||||
FROM nginx:${NGINX_VERSION} AS runtime
|
||||
|
||||
# Drop the upstream default :80 server; we own the config.
|
||||
RUN rm -f /etc/nginx/conf.d/default.conf
|
||||
|
||||
# Static export → web root.
|
||||
COPY --from=builder /app/out /usr/share/nginx/html
|
||||
|
||||
# Routing rules — see ui/nginx.conf for the full description.
|
||||
COPY ui/nginx.conf /etc/nginx/nginx.conf
|
||||
|
||||
EXPOSE 3000/tcp
|
||||
|
||||
# nginx as PID 1 in foreground; respects SIGTERM out of the box, so
|
||||
# no tini/dumb-init wrapper needed.
|
||||
CMD ["nginx", "-g", "daemon off;"]
|
||||
|
|
@ -5,6 +5,7 @@ worker_processes auto;
|
|||
# nginx image's /var/cache/nginx and /run are root-owned 755) and works
|
||||
# with readOnlyRootFilesystem when /tmp is an emptyDir.
|
||||
pid /tmp/nginx.pid;
|
||||
error_log /dev/stderr warn;
|
||||
|
||||
events { worker_connections 1024; }
|
||||
|
||||
|
|
@ -15,6 +16,7 @@ http {
|
|||
uwsgi_temp_path /tmp/nginx-uwsgi-temp;
|
||||
scgi_temp_path /tmp/nginx-scgi-temp;
|
||||
|
||||
access_log /dev/stdout;
|
||||
include /etc/nginx/mime.types;
|
||||
default_type application/octet-stream;
|
||||
sendfile on;
|
||||
|
|
@ -29,7 +31,6 @@ http {
|
|||
application/javascript
|
||||
application/json
|
||||
text/css
|
||||
text/html
|
||||
image/svg+xml
|
||||
font/woff
|
||||
font/woff2;
|
||||
|
|
@ -37,16 +38,16 @@ http {
|
|||
server {
|
||||
listen 3000 default_server;
|
||||
server_name _;
|
||||
root /usr/share/nginx/html;
|
||||
root /var/lib/litellm/ui;
|
||||
|
||||
# next.config.mjs sets assetPrefix=/litellm-asset-prefix, which makes
|
||||
# the built HTML reference /litellm-asset-prefix/_next/... — but the
|
||||
# static export only emits files under /_next/. Map the prefix to
|
||||
# the real tree at request time instead of duplicating the directory
|
||||
# at build time. NB: alias rewrites the location prefix, so
|
||||
# /litellm-asset-prefix/_next/foo.js → /usr/share/nginx/html/_next/foo.js.
|
||||
# /litellm-asset-prefix/_next/foo.js → /var/lib/litellm/ui/_next/foo.js.
|
||||
location /litellm-asset-prefix/_next/ {
|
||||
alias /usr/share/nginx/html/_next/;
|
||||
alias /var/lib/litellm/ui/_next/;
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
|
|
@ -63,7 +64,7 @@ http {
|
|||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
location ^~ /ui/assets/ {
|
||||
alias /usr/share/nginx/html/assets/;
|
||||
alias /var/lib/litellm/ui/assets/;
|
||||
}
|
||||
location = /favicon.ico {
|
||||
try_files $uri =404;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue