diff --git a/.circleci/config.yml b/.circleci/config.yml index 7276da9877b..9b1e433be93 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -532,11 +532,10 @@ jobs: key: v1-uv-cache-{{ checksum "uv.lock" }} - save_cargo_target - run: - name: Run prisma ./docker/entrypoint.sh + name: Run prisma migrations command: | set +e - chmod +x docker/entrypoint.sh - ./docker/entrypoint.sh + uv run --no-sync python litellm/proxy/prisma_migration.py set -e # Run pytest and generate JUnit XML report - run: @@ -606,11 +605,10 @@ jobs: - ~/.cache/uv key: v1-uv-cache-{{ checksum "uv.lock" }} - run: - name: Run prisma ./docker/entrypoint.sh + name: Run prisma migrations command: | set +e - chmod +x docker/entrypoint.sh - ./docker/entrypoint.sh + uv run --no-sync python litellm/proxy/prisma_migration.py set -e # Run pytest and generate JUnit XML report - run: @@ -681,11 +679,10 @@ jobs: - ~/.cache/uv key: v1-uv-cache-{{ checksum "uv.lock" }} - run: - name: Run prisma ./docker/entrypoint.sh + name: Run prisma migrations command: | set +e - chmod +x docker/entrypoint.sh - ./docker/entrypoint.sh + uv run --no-sync python litellm/proxy/prisma_migration.py set -e # Run pytest and generate JUnit XML report @@ -2422,16 +2419,18 @@ jobs: path: test-results proxy_build_from_pip_tests: - # Change from docker to machine executor + # Validates the published PyPI artifact, not the checked-out source tree: the + # proxy is pip-installed from PyPI into its own venv and booted on the runner machine: image: ubuntu-2204:2024.04.1 resource_class: large working_directory: ~/project + environment: + LITELLM_PIP_VERSION: "1.83.0" steps: - checkout - skip_if_unrelated_changes - setup_google_dns - # Remove Docker CLI installation since it's already available in machine executor - install_uv - install_rust - run: @@ -2439,47 +2438,53 @@ jobs: command: | uv sync --frozen --all-groups --all-extras --python 3.12 - run: - name: Build Docker image + name: Install the published litellm proxy from PyPI into its own venv command: | - docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip . + uv venv --python 3.13 /tmp/litellm-pip + uv pip install --python /tmp/litellm-pip/bin/python \ + "litellm[proxy,proxy-runtime]==${LITELLM_PIP_VERSION}" \ + "google-cloud-aiplatform==1.133.0" \ + "google-genai==1.37.0" \ + "anthropic[vertex]==0.84.0" \ + "grpcio==1.78.0" \ + "prometheus-client==0.20.0" \ + "langfuse==2.59.7" \ + "opentelemetry-api==1.28.0" \ + "opentelemetry-sdk==1.28.0" \ + "opentelemetry-exporter-otlp==1.28.0" \ + "ddtrace==4.11.0" \ + "sentry-sdk==2.21.0" \ + "mangum==0.17.0" \ + "azure-ai-contentsafety==1.0.0" \ + "azure-storage-file-datalake==12.20.0" \ + "pypdf==6.7.5" \ + "llm-sandbox==0.3.31" \ + "detect-secrets==1.5.0" \ + "prisma==0.11.0" \ + "openai==2.24.0" + /tmp/litellm-pip/bin/python -c "import litellm, sys; print('litellm', litellm.__version__ if hasattr(litellm, '__version__') else '', sys.version)" - start_postgres - start_fake_openai_endpoint - run: - name: Run Docker container + name: Run the published proxy # intentionally give bad redis credentials here # the OTEL test - should get this as a trace - command: | - docker run -d \ - -p 4000:4000 \ - -e LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true \ - -e DATABASE_URL=postgresql://postgres:postgres@host.docker.internal:5432/circle_test \ - -e REDIS_HOST=$REDIS_HOST \ - -e REDIS_PASSWORD=$REDIS_PASSWORD \ - -e REDIS_PORT=$REDIS_PORT \ - -e LITELLM_MASTER_KEY="sk-1234" \ - -e OPENAI_API_KEY=$OPENAI_API_KEY \ - -e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \ - -e LITELLM_LICENSE=$LITELLM_LICENSE \ - -e OTEL_EXPORTER="in_memory" \ - -e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \ - -e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \ - -e AWS_REGION_NAME=$AWS_REGION_NAME \ - -e COHERE_API_KEY=$COHERE_API_KEY \ - -e USE_DDTRACE=True \ - -e DD_API_KEY=$DD_API_KEY \ - -e DD_SITE=$DD_SITE \ - -e GCS_FLUSH_INTERVAL="1" \ - -e LITELLM_LOG=ERROR \ - --add-host host.docker.internal:host-gateway \ - --name my-app \ - -v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \ - my-app:latest \ - --config /app/config.yaml \ - --port 4000 - - run: - name: Start outputting logs - command: docker logs -f my-app background: true + command: | + export LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY=true + export DATABASE_URL=postgresql://postgres:postgres@localhost:5432/circle_test + export LITELLM_MASTER_KEY="sk-1234" + export LITELLM_LICENSE="bad-license" + export FAKE_OPENAI_API_BASE=http://localhost:8190 + export OTEL_EXPORTER="in_memory" + export USE_DDTRACE=True + export DD_TRACE_OPENAI_ENABLED="False" + export GCS_FLUSH_INTERVAL="1" + export LITELLM_LOG=ERROR + cd /tmp/litellm-pip + exec /tmp/litellm-pip/bin/ddtrace-run /tmp/litellm-pip/bin/litellm \ + --config ~/project/tests/basic_proxy_startup_tests/build_from_pip_config.yaml \ + --port 4000 2>&1 | tee /tmp/litellm-pip.log - wait_for_service: url: http://localhost:4000 timeout: "300" @@ -2495,14 +2500,15 @@ jobs: --junitxml=test-results/junit-2.xml \ --durations=5" no_output_timeout: 15m - # Clean up first container - store_test_results: path: test-results - run: - name: Stop and remove first container + name: Proxy log + command: cat /tmp/litellm-pip.log || true + when: always + - run: + name: Stop postgres command: | - docker stop my-app || true - docker rm my-app || true docker stop postgres-db || true docker rm postgres-db || true when: always @@ -3023,7 +3029,7 @@ jobs: docker build \ --label org.opencontainers.image.revision="$(git rev-parse HEAD)" \ -t litellm-docker-database:ci \ - -f docker/Dockerfile.database . + -f Dockerfile . fi python3 .circleci/scripts/run_migration_tests.py record-image diff --git a/.dockerignore b/.dockerignore index f3a80fee3e4..8459d0cd482 100644 --- a/.dockerignore +++ b/.dockerignore @@ -7,7 +7,6 @@ tests .devcontainer *.tgz log.txt -docker/Dockerfile.* # Claude Flow generated files (must be excluded from Docker build) .claude/ diff --git a/.github/ci-coverage-allowlist.yml b/.github/ci-coverage-allowlist.yml index 445a8519436..d9ba8dd1d0e 100644 --- a/.github/ci-coverage-allowlist.yml +++ b/.github/ci-coverage-allowlist.yml @@ -102,15 +102,6 @@ test_paths: - tests/integration/test_oci_proxy_integration.py dockerfiles: - - reason: >- - The dashboard container is a static Next.js export served by nginx, and the dashboard build - and lint workflows already exercise that output, so building the image adds no signal about it - paths: - - ui/Dockerfile - - reason: >- - An example image under cookbook/ that is documentation rather than a shipped artifact - paths: - - cookbook/litellm-ollama-docker-image/Dockerfile - reason: >- The Rust gateway image compiles the whole workspace in release mode, which is too slow for a per-pull-request job while the gateway binary is still being assembled; the Rust lint, diff --git a/.github/workflows/compat-matrix-image.yml b/.github/workflows/compat-matrix-image.yml index f08792c904c..8d68bb1213c 100644 --- a/.github/workflows/compat-matrix-image.yml +++ b/.github/workflows/compat-matrix-image.yml @@ -26,17 +26,24 @@ jobs: with: persist-credentials: false - - name: Build the Render cron image - run: docker build -f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix:${{ github.sha }} tests/e2e - - - name: Resolve and install the Claude Code CLI as the cron user + - name: Install the pinned uv the cron image ships + env: + UV_VERSION: 0.10.9 + UV_SHA256: 20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594 run: | - docker run --rm compat-matrix:${{ github.sha }} bash -c ' - set -euo pipefail - whoami - gh --version - uv --version - version="$(uv run --no-project --python 3.12 python /opt/litellm/tests/e2e/claude_code/pr_gate_version_resolver.py)" - /opt/litellm/tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" /tmp/claude-cli - /tmp/claude-cli/claude --version - ' + set -euo pipefail + curl -fsSLo "$RUNNER_TEMP/uv.tar.gz" "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" + echo "${UV_SHA256} $RUNNER_TEMP/uv.tar.gz" | sha256sum -c - + mkdir -p "$RUNNER_TEMP/bin" + tar -xzf "$RUNNER_TEMP/uv.tar.gz" -C "$RUNNER_TEMP/bin" --strip-components=1 uv-x86_64-unknown-linux-gnu/uv + echo "$RUNNER_TEMP/bin" >> "$GITHUB_PATH" + + - name: Resolve and install the Claude Code CLI as an unprivileged user + run: | + set -euo pipefail + whoami + gh --version + uv --version + version="$(uv run --no-project --python 3.12 python tests/e2e/claude_code/pr_gate_version_resolver.py)" + tests/e2e/claude_code/cron_vm/install_claude_code.sh "${version}" "$RUNNER_TEMP/claude-cli" + "$RUNNER_TEMP/claude-cli/claude" --version diff --git a/.github/workflows/image-scan.yml b/.github/workflows/image-scan.yml index 0695720733f..a08ea18d05c 100644 --- a/.github/workflows/image-scan.yml +++ b/.github/workflows/image-scan.yml @@ -8,21 +8,17 @@ on: - "litellm_**" paths: - Dockerfile - - docker/Dockerfile.non_root - - migrations/Dockerfile + - .dockerignore + - docker-entrypoint.sh - migrations/run.py - - gateway/Dockerfile - gateway/main.py - - backend/Dockerfile + - gateway/launch.py - backend/main.py - - docker/component_entrypoint.sh - - docker/entrypoint.sh - litellm/proxy/prisma_migration.py - litellm-proxy-extras/** - tests/proxy_migration_tests/** - uv.lock - ui/litellm-dashboard/package-lock.json - - ui/Dockerfile - ui/nginx.conf - .github/workflows/image-scan.yml - .grype.yaml @@ -36,9 +32,12 @@ concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true +env: + IMAGE: litellm-image-scan:${{ github.sha }} + jobs: - image-scan: - name: image-scan + build: + name: build runs-on: ubuntu-latest if: >- github.event_name != 'pull_request' || @@ -51,7 +50,92 @@ jobs: with: persist-credentials: false + # One image ships; every component below is a mode of it. Built once + # here and handed to the verify matrix so a component check never runs + # against a different build than the scan. + - name: Build the image + run: docker build -t "$IMAGE" . + + - name: Save the image for the verify matrix + run: docker save "$IMAGE" | zstd -T0 -3 -o "$RUNNER_TEMP/litellm-image.tar.zst" + + - uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1 + with: + name: litellm-image + path: ${{ runner.temp }}/litellm-image.tar.zst + retention-days: 1 + compression-level: 0 + + verify: + name: ${{ matrix.check }} + needs: build + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + contents: read + strategy: + fail-fast: false + matrix: + check: [scan, migrations, gateway, backend, ui] + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 + with: + name: litellm-image + path: ${{ runner.temp }} + + - name: Load the image + run: zstd -d --stdout "$RUNNER_TEMP/litellm-image.tar.zst" | docker load + + - name: Set up Python + if: matrix.check != 'scan' + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Install pytest + if: matrix.check != 'scan' + run: python -m pip install "pytest==9.0.3" + + # The prisma bake must migrate a fresh DB with no egress as an arbitrary + # non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyFilesystem). + # `docker run` as the default uid with network hides a broken bake because + # the migration entrypoint exits 0 even when it applied nothing; asserting + # the schema was created is what catches it. + - name: Verify offline migration as a non-root uid + if: matrix.check == 'migrations' + env: + LITELLM_IMAGE: ${{ env.IMAGE }} + run: | + python -m pytest -v \ + tests/proxy_migration_tests/test_offline_image_migration.py \ + tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py + + - name: Verify the gateway component serves offline as a non-root uid + if: matrix.check == 'gateway' + env: + LITELLM_IMAGE: ${{ env.IMAGE }} + LITELLM_COMPONENT: gateway + run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py + + - name: Verify the backend component serves offline as a non-root uid + if: matrix.check == 'backend' + env: + LITELLM_IMAGE: ${{ env.IMAGE }} + LITELLM_COMPONENT: backend + run: python -m pytest -v tests/proxy_migration_tests/test_component_image_serves_offline.py + + - name: Verify the ui component serves as an arbitrary uid with a read-only root fs + if: matrix.check == 'ui' + env: + LITELLM_IMAGE: ${{ env.IMAGE }} + run: python -m pytest -v tests/proxy_migration_tests/test_ui_image_serves_offline.py + - name: Download Grype v0.114.0 + if: matrix.check == 'scan' run: | curl -fsSL --retry 3 -o "$RUNNER_TEMP/grype.tar.gz" \ https://github.com/anchore/grype/releases/download/v0.114.0/grype_0.114.0_linux_amd64.tar.gz @@ -59,29 +143,6 @@ jobs: tar xzf "$RUNNER_TEMP/grype.tar.gz" -C "$RUNNER_TEMP" grype chmod +x "$RUNNER_TEMP/grype" - # Dockerfile.non_root is the rootless variant we ship. The other - # Dockerfiles share the same wolfi base and apk set, so OS-layer coverage - # is the same; matrix-scan if those variants ever diverge. - - name: Build runtime image - run: docker build -f docker/Dockerfile.non_root -t litellm-image-scan:${{ github.sha }} . - - # The prisma bake must migrate a fresh DB with no egress as an arbitrary - # non-root uid (OpenShift restricted-v2 / air-gapped / readOnlyRootFilesystem). - # `docker run` as the default uid with network hides a broken bake because - # the migration entrypoint exits 0 even when it applied nothing; asserting - # the schema was created is what catches it. - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify offline migration as a non-root uid - env: - LITELLM_IMAGE: litellm-image-scan:${{ github.sha }} - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v - # Scans the whole shipped artifact: OS/apk plus every language package # baked into the image, including ones no lockfile declares (e.g. prisma's # vendored node engine) that osv-scan cannot see. osv-scan stays the fast @@ -89,160 +150,12 @@ jobs: # free OSS, run as a pinned, checksum-verified binary; no GitHub Action # dependency and no vendor SaaS callout. - name: Scan image for fixable HIGH/CRITICAL CVEs + if: matrix.check == 'scan' env: GRYPE_MATCH_PYTHON_USING_CPES: "true" run: | - "$RUNNER_TEMP/grype" litellm-image-scan:${{ github.sha }} \ + "$RUNNER_TEMP/grype" "$IMAGE" \ --config .grype.yaml \ --only-fixed \ --fail-on high \ --output table - - runtime-image: - name: runtime-image - runs-on: ubuntu-latest - if: >- - github.event_name != 'pull_request' || - github.event.pull_request.head.repo.full_name == github.repository - timeout-minutes: 30 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build runtime image - run: docker build -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} . - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify offline migration as a non-root uid - env: - LITELLM_IMAGE: litellm-runtime-scan:${{ github.sha }} - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v - - migrations-image: - name: migrations-image - runs-on: ubuntu-latest - if: >- - github.event_name != 'pull_request' || - github.event.pull_request.head.repo.full_name == github.repository - timeout-minutes: 30 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build migrations image - run: docker build -f migrations/Dockerfile -t litellm-migrations-scan:${{ github.sha }} . - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify offline migration as a non-root uid - env: - LITELLM_IMAGE: litellm-migrations-scan:${{ github.sha }} - LITELLM_MIGRATION_INTERPRETER: python3 - LITELLM_MIGRATION_SCRIPT: /app/run.py - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py -v - - gateway-image: - name: gateway-image - runs-on: ubuntu-latest - if: >- - github.event_name != 'pull_request' || - github.event.pull_request.head.repo.full_name == github.repository - timeout-minutes: 30 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build gateway image - run: docker build -f gateway/Dockerfile -t litellm-gateway-scan:${{ github.sha }} . - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify the gateway serves offline as a non-root uid - env: - LITELLM_IMAGE: litellm-gateway-scan:${{ github.sha }} - LITELLM_COMPONENT_PORT: "4000" - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v - - ui-image: - name: ui-image - runs-on: ubuntu-latest - if: >- - github.event_name != 'pull_request' || - github.event.pull_request.head.repo.full_name == github.repository - timeout-minutes: 30 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build UI image - run: docker build -f ui/Dockerfile -t litellm-ui-scan:${{ github.sha }} . - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify the UI serves offline as an arbitrary uid with a read-only root fs - env: - LITELLM_IMAGE: litellm-ui-scan:${{ github.sha }} - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_ui_image_serves_offline.py -v - - backend-image: - name: backend-image - runs-on: ubuntu-latest - if: >- - github.event_name != 'pull_request' || - github.event.pull_request.head.repo.full_name == github.repository - timeout-minutes: 30 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Build backend image - run: docker build -f backend/Dockerfile -t litellm-backend-scan:${{ github.sha }} . - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Verify the backend serves offline as a non-root uid - env: - LITELLM_IMAGE: litellm-backend-scan:${{ github.sha }} - LITELLM_COMPONENT_PORT: "4001" - run: | - python -m pip install "pytest==9.0.3" - python -m pytest tests/proxy_migration_tests/test_component_image_serves_offline.py -v diff --git a/.github/workflows/test-litellm-ui-build.yml b/.github/workflows/test-litellm-ui-build.yml index eace78fc2cb..6ffb1a793fe 100644 --- a/.github/workflows/test-litellm-ui-build.yml +++ b/.github/workflows/test-litellm-ui-build.yml @@ -33,8 +33,8 @@ jobs: # Built through the image stage rather than the checkout, because the # stage copies ui/litellm-dashboard/ alone: an import reaching above the # dashboard root resolves in a checkout and fails in every image we ship. - # Dockerfile, docker/Dockerfile.non_root and ui/Dockerfile share this - # stage verbatim, so building one covers all three. + # The shipped image serves this stage's output in every component, so + # building the stage here covers what ships. - name: Build the dashboard as the shipped images build it if: steps.changes.outputs.decision != 'skip' run: docker build --target ui-builder -f Dockerfile . diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a5ad6e97f3d..7017e7e3890 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -265,8 +265,8 @@ uv run litellm --config your_config.yaml If you want to build the Docker image yourself: ```bash -# Build using the non-root Dockerfile -docker build -f docker/Dockerfile.non_root -t litellm_dev . +# Build the image (runs as a non-root user by default) +docker build -t litellm_dev . # Generate a master key. Requests send it as the bearer token export LITELLM_MASTER_KEY="sk-$(openssl rand -hex 32)" diff --git a/Dockerfile b/Dockerfile index 4dcecf3ea3d..6599c5a4478 100644 --- a/Dockerfile +++ b/Dockerfile @@ -111,8 +111,12 @@ RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/ npm_config_cache=/root/.npm \ prisma generate --schema=./schema.prisma -RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \ - sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh +RUN sed -i 's/\r$//' docker-entrypoint.sh && chmod +x docker-entrypoint.sh + +RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \ + cp -r litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \ + cp litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \ + touch /var/lib/litellm/ui/.litellm_ui_ready # Runtime stage FROM $LITELLM_RUNTIME_IMAGE AS runtime @@ -125,23 +129,40 @@ USER root # https://github.com/BerriAI/litellm/issues/33518 RUN echo "https://packages.wolfi.dev/os" >> /etc/apk/repositories -# node (without npm) is required by the prisma CLI at runtime -RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent +# node (without npm) is required by the prisma CLI at runtime; nginx serves the +# static admin UI in the `ui` component; libevent is pgbouncer's runtime. +RUN for i in 1 2 3; do \ + apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libatomic libevent nginx && break; \ + [ "$i" = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ + sleep 5; \ + done COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer WORKDIR /app + +# Runtime writes land under /app/.cache, /var/lib/litellm and /tmp (group-0 +# writable, so an arbitrary uid and a read-only root fs both work). Prisma CLI +# and engines are baked under /opt/prisma so `prisma migrate deploy` needs no +# npm and no network (#33650, #24554). ENV PATH="/app/.venv/bin:${PATH}" \ + PYTHONPATH="/app" \ + PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + HOME=/app \ + LITELLM_NON_ROOT=true \ PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \ PRISMA_CLI_QUERY_ENGINE_TYPE=binary \ + PRISMA_SKIP_POSTINSTALL_GENERATE=1 \ + PRISMA_HIDE_UPDATE_MESSAGE=1 \ + PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \ PRISMA_OFFLINE_MODE=true # Copy only what runtime needs. The application is installed inside the venv; # the rest of the builder's /app is source and build metadata that must not # ship (manifest-scanning tools attribute everything in it to this image). -# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path. COPY --from=builder /app/.venv /app/.venv -COPY --from=builder /app/docker /app/docker +COPY --from=builder /app/docker-entrypoint.sh /app/docker-entrypoint.sh COPY --from=builder /app/schema.prisma /app/schema.prisma COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py # enterprise/ is imported by source path at runtime (proxy_cli puts the @@ -149,21 +170,34 @@ COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/pr # enterprise.enterprise_hooks from it) COPY --from=builder /app/enterprise /app/enterprise COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras -# Prisma CLI + engines are baked under /opt/prisma, a fixed path every -# runtime uid can read and that no cache volume mount shadows. The paths are -# pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and recorded into the -# generated client at build time, so `prisma migrate deploy` on a fresh -# database needs no npm and no network access (#33650, #24554). +COPY --from=builder /app/gateway /app/gateway +COPY --from=builder /app/backend /app/backend +COPY --from=builder /app/migrations /app/migrations COPY --from=builder /opt/prisma /opt/prisma +COPY --from=builder /var/lib/litellm /var/lib/litellm +COPY ui/nginx.conf /etc/nginx/nginx.conf -RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ - find /app/.venv -type d -path "*/tornado/test" -delete && \ +RUN find /app/.venv -depth -type d -path "*/tornado/test" -exec rm -rf {} + && \ + mkdir -p /app/.cache && \ + chown -R 65532:0 /app /var/lib/litellm && \ + chmod -R g=u,g+w /app/.cache /var/lib/litellm && \ + PRISMA_PATH="$(python -c 'import os, prisma; print(os.path.dirname(prisma.__file__))')" && \ + PROXY_EXTRAS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))')" && \ + chmod -R g=u,g+w "$PRISMA_PATH" "$PROXY_EXTRAS_PATH" && \ chmod -R a+rX /opt/prisma && \ test -x /opt/prisma/binaries/node_modules/.bin/prisma && \ test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \ - python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" + ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null && \ + python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \ + python -c "from litellm.rust_bridge.loader import native_bridge_available; assert native_bridge_available()" && \ + python -c "import gateway.launch, backend.main" -EXPOSE 4000/tcp +# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The +# numeric form is what the kubelet's runAsNonRoot admission check can verify. +USER 65532:65532 -ENTRYPOINT ["docker/prod_entrypoint.sh"] -CMD ["--port", "4000"] +RUN nginx -t + +EXPOSE 4000/tcp 4001/tcp 3000/tcp + +ENTRYPOINT ["/app/docker-entrypoint.sh"] diff --git a/backend/Dockerfile b/backend/Dockerfile deleted file mode 100644 index 59f836b55f8..00000000000 --- a/backend/Dockerfile +++ /dev/null @@ -1,106 +0,0 @@ -ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a - -FROM $UV_IMAGE AS uvbin - -# ---------- Builder ---------- -FROM $LITELLM_BUILD_IMAGE AS builder - -WORKDIR /app -USER root - -COPY --from=uvbin /uv /uvx /usr/local/bin/ - -# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE -# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi -# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes. -RUN for i in 1 2 3; do \ - apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start. -# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a -# BuildKit cache mount (different filesystem). -# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of -# silently pulling a managed interpreter. -# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't -# silently re-enable nodeenv's Node download. -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - UV_COMPILE_BYTECODE=1 \ - UV_PYTHON_DOWNLOADS=0 \ - PRISMA_USE_GLOBAL_NODE=true \ - PATH="/app/.venv/bin:${PATH}" - -# Stage 1 — install dependencies only. -RUN --mount=type=cache,target=/root/.cache/uv \ - --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ - --mount=type=bind,source=uv.lock,target=uv.lock \ - --mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \ - --mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \ - uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --python python3.13 - -# Stage 2 — copy source and install the project + workspace members. -COPY . . - -RUN --mount=type=cache,target=/root/.cache/uv \ - uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --python python3.13 - -RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/ - -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma - -RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh - -# ---------- Runtime ---------- -FROM $LITELLM_RUNTIME_IMAGE AS runtime - -USER root - -RUN for i in 1 2 3; do \ - apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with -# /home/nonroot. We run the backend as that user -WORKDIR /app -ENV HOME=/home/nonroot \ - PATH="/app/.venv/bin:${PATH}" \ - PYTHONPATH="/app" \ - PYTHONDONTWRITEBYTECODE=1 \ - PYTHONUNBUFFERED=1 \ - PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries - -COPY --from=builder --chown=nonroot:nonroot /app /app -COPY --from=builder /opt/prisma /opt/prisma - -RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ - find /app/.venv -type d -path "*/tornado/test" -delete && \ - chmod -R a+rX /opt/prisma && \ - python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" - -USER nonroot - -EXPOSE 4001/tcp - -ENTRYPOINT ["/app/docker/component_entrypoint.sh", "uvicorn", "backend.main:app"] -CMD ["--host", "0.0.0.0", "--port", "4001"] diff --git a/ci_cd/security_scans_readme.md b/ci_cd/security_scans_readme.md index dd64b01c296..3003ea1d105 100644 --- a/ci_cd/security_scans_readme.md +++ b/ci_cd/security_scans_readme.md @@ -4,6 +4,5 @@ - Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM) - Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM) -- Grype scan on `Dockerfile.database` (fails on CRITICAL) - Grype scan on main `Dockerfile` (fails on CRITICAL) - Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0) diff --git a/cookbook/litellm-ollama-docker-image/Dockerfile b/cookbook/litellm-ollama-docker-image/Dockerfile deleted file mode 100644 index be237a4df77..00000000000 --- a/cookbook/litellm-ollama-docker-image/Dockerfile +++ /dev/null @@ -1,25 +0,0 @@ -FROM ollama/ollama as ollama - -RUN echo "auto installing llama2" - -# auto install ollama/llama2 -RUN ollama serve & sleep 2 && ollama pull llama2 - -RUN echo "installing litellm" - -RUN apt-get update - -# Install Python -RUN apt-get install -y python3 python3-pip - -# Set the working directory in the container -WORKDIR /app - -# Copy the current directory contents into the container at /app -COPY . /app - -# Install any needed packages specified in requirements.txt - -RUN python3 -m pip install litellm -COPY start.sh /start.sh -ENTRYPOINT [ "/bin/bash", "/start.sh" ] diff --git a/cookbook/litellm-ollama-docker-image/requirements.txt b/cookbook/litellm-ollama-docker-image/requirements.txt deleted file mode 100644 index 9b9181b2360..00000000000 --- a/cookbook/litellm-ollama-docker-image/requirements.txt +++ /dev/null @@ -1 +0,0 @@ -litellm==1.83.14 diff --git a/cookbook/litellm-ollama-docker-image/start.sh b/cookbook/litellm-ollama-docker-image/start.sh deleted file mode 100644 index ecc03ce73c5..00000000000 --- a/cookbook/litellm-ollama-docker-image/start.sh +++ /dev/null @@ -1,2 +0,0 @@ -ollama serve & -litellm \ No newline at end of file diff --git a/cookbook/litellm-ollama-docker-image/test.py b/cookbook/litellm-ollama-docker-image/test.py deleted file mode 100644 index 93b9c6ac4a7..00000000000 --- a/cookbook/litellm-ollama-docker-image/test.py +++ /dev/null @@ -1,35 +0,0 @@ -import openai - -api_base = "http://0.0.0.0:8000" - -openai.api_base = api_base -openai.api_key = "temp-key" -print(openai.api_base) - - -print("LiteLLM: response from proxy with streaming") -response = openai.ChatCompletion.create( - model="ollama/llama2", - messages=[ - { - "role": "user", - "content": "this is a test request, acknowledge that you got it", - } - ], - stream=True, -) - -for chunk in response: - print(f"LiteLLM: streaming response from proxy {chunk}") - -response = openai.ChatCompletion.create( - model="ollama/llama2", - messages=[ - { - "role": "user", - "content": "this is a test request, acknowledge that you got it", - } - ], -) - -print(f"LiteLLM: response from proxy {response}") diff --git a/docker-compose.hardened.yml b/docker-compose.hardened.yml deleted file mode 100644 index 31d0c2e9ef2..00000000000 --- a/docker-compose.hardened.yml +++ /dev/null @@ -1,46 +0,0 @@ -services: - # Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints. - # Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage. - litellm: - build: - context: . - dockerfile: docker/Dockerfile.non_root - target: runtime - args: - PROXY_EXTRAS_SOURCE: "local" - depends_on: - - squid - user: "101:101" - group_add: - - "2345" - read_only: true - cap_drop: - - ALL - security_opt: - - no-new-privileges:true - tmpfs: - - /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777 - - /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777 - volumes: - - ./proxy_server_config.yaml:/app/config.yaml:ro - environment: - LITELLM_NON_ROOT: "true" - PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries" - XDG_CACHE_HOME: "/app/cache" - LITELLM_MIGRATION_DIR: "/app/migrations" - HTTP_PROXY: "http://squid:3128" - HTTPS_PROXY: "http://squid:3128" - NO_PROXY: "localhost,127.0.0.1,db" - command: - - "--port" - - "4000" - - "--config" - - "/app/config.yaml" - squid: - image: sameersbn/squid:3.5.27-2 - restart: unless-stopped - ports: - - "3128:3128" - tmpfs: - - /var/spool/squid:rw,noexec,nosuid,nodev,size=64m - - /var/log/squid:rw,noexec,nosuid,nodev,size=16m diff --git a/docker-compose.yml b/docker-compose.yml index 80e1f289aad..a419c053b7e 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,52 +1,69 @@ +# LiteLLM plus a Postgres database that stores models, virtual keys and spend +# logs. Used by https://docs.litellm.ai/docs/proxy/docker_quick_start +# +# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml +# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env +# docker compose up -d +# +# `docker compose up` runs the published image. From a checkout of the repo, +# `docker compose up --build` builds the Dockerfile next to this file instead. +# `docker compose --profile monitoring up -d` also starts Prometheus on 9090. +# +# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY +# makes credentials already stored in the database unreadable. For anything +# beyond local evaluation, pin the image to a specific release tag. services: litellm: + image: docker.litellm.ai/berriai/litellm:main-stable build: context: . - args: - target: runtime - image: docker.litellm.ai/berriai/litellm:main-stable - ######################################### - ## Uncomment these lines to start proxy with a config.yaml file ## - # volumes: - # - ./config.yaml:/app/config.yaml - # command: - # - "--config=/app/config.yaml" - ############################################## + target: runtime + pull_policy: missing ports: - - "4000:4000" # Map the container port to the host, change the host port if necessary + - "4000:4000" + # The same image runs every component. The first word of `command` picks + # it: proxy (default), gateway, backend, ui, migrations, metrics, + # collector. See docker/README.md + # command: ["--config", "/app/config.yaml"] + # volumes: + # - ./config.yaml:/app/config.yaml:ro environment: - DATABASE_URL: "postgresql://llmproxy:dbpassword9090@db:5432/litellm" - # Optional: route read-only queries (find_*, count, group_by, query_raw/_first) - # to a separate reader endpoint, e.g. an Aurora reader. Leave unset for - # single-DB deployments. With IAM_TOKEN_DB_AUTH enabled, the reader URL - # is auto-refreshed alongside the writer. - # DATABASE_URL_READ_REPLICA: "postgresql://llmproxy:dbpassword9090@db-reader:5432/litellm" - STORE_MODEL_IN_DB: "True" # allows adding models to proxy via UI + LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env, see the header of this file} + LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env, see the header of this file} + DATABASE_URL: postgresql://llmproxy:dbpassword9090@db:5432/litellm + STORE_MODEL_IN_DB: "True" env_file: - - .env # Load local .env file + - path: .env + required: false + read_only: true + cap_drop: + - ALL + security_opt: + - no-new-privileges:true + tmpfs: + - /tmp:rw,noexec,nosuid,nodev,size=64m,mode=1777 + - /app/.cache:rw,noexec,nosuid,nodev,size=128m,mode=1777 depends_on: - - db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first - healthcheck: # Defines the health check configuration for the container + db: + condition: service_healthy + healthcheck: test: - CMD-SHELL - - python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')" # Command to execute for health check - interval: 30s # Perform health check every 30 seconds - timeout: 10s # Health check command times out after 10 seconds - retries: 3 # Retry up to 3 times if health check fails - start_period: 40s # Wait 40 seconds after container start before beginning health checks + - python3 -c "import urllib.request; urllib.request.urlopen('http://localhost:4000/health/liveliness')" + interval: 30s + timeout: 10s + retries: 3 + start_period: 40s db: image: postgres:16 restart: always - container_name: litellm_db environment: POSTGRES_DB: litellm POSTGRES_USER: llmproxy POSTGRES_PASSWORD: dbpassword9090 - ports: - - "5432:5432" volumes: - - postgres_data:/var/lib/postgresql/data # Persists Postgres data across container restarts + - postgres_data:/var/lib/postgresql/data healthcheck: test: ["CMD-SHELL", "pg_isready -d litellm -U llmproxy"] interval: 1s @@ -55,6 +72,8 @@ services: prometheus: image: prom/prometheus + profiles: + - monitoring volumes: - prometheus_data:/prometheus - ./prometheus.yml:/etc/prometheus/prometheus.yml @@ -68,6 +87,5 @@ services: volumes: prometheus_data: - driver: local postgres_data: - name: litellm_postgres_data # Named volume for Postgres data persistence + name: litellm_postgres_data diff --git a/docker-entrypoint.sh b/docker-entrypoint.sh new file mode 100755 index 00000000000..48bb6b582ce --- /dev/null +++ b/docker-entrypoint.sh @@ -0,0 +1,93 @@ +#!/bin/sh +# LiteLLM image entrypoint: docker-entrypoint.sh [COMPONENT] [ARGS...] +# +# COMPONENT selects the process this container runs; it can also be given as +# LITELLM_COMPONENT when the command carries no component. Anything after it is +# passed to that process. A first argument starting with "-" (or no argument at +# all) keeps the historical behaviour of running the monolithic proxy. +# +# proxy litellm ARGS everything in one process (default) +# gateway python -m gateway.launch ARGS inference routes, 0.0.0.0:4000 +# backend uvicorn backend.main:app ARGS management routes, 0.0.0.0:4001 +# ui nginx serving the admin UI port 3000 +# migrations python migrations/run.py prisma migrate deploy, then exit +# metrics python -m litellm.proxy.prometheus_metrics_server ARGS +# collector python -m litellm.proxy.collector ARGS +# +# gateway and backend get their host and port defaults first, so ARGS such as +# --port 8080 override them. An unknown first word is executed as-is (docker run +# sh). PgBouncer is not a component: LITELLM_PGBOUNCER_ENABLED=true +# starts it inside proxy and gateway. Components that write Prometheus samples +# start with an empty PROMETHEUS_MULTIPROC_DIR; metrics, collector and raw +# commands get the directory created but keep the workers' files. +set -eu + +usage() { + sed -n '2,24p' "$0" | sed 's/^# \{0,1\}//' >&2 +} + +component="${LITELLM_COMPONENT:-proxy}" +case "${1:-}" in + proxy|gateway|backend|ui|migrations|metrics|collector) + component="$1" + shift + ;; + ""|-*) + ;; + help) + usage + exit 0 + ;; + *) + [ -z "${PROMETHEUS_MULTIPROC_DIR:-}" ] || mkdir -p "$PROMETHEUS_MULTIPROC_DIR" + exec "$@" + ;; +esac + +case "$component" in + proxy) + set -- litellm "$@" + ;; + gateway) + set -- python -m gateway.launch --workers "${NUM_WORKERS:-1}" --host 0.0.0.0 --port 4000 "$@" + ;; + backend) + set -- uvicorn backend.main:app --host 0.0.0.0 --port 4001 "$@" + ;; + ui) + exec nginx -g 'daemon off;' "$@" + ;; + migrations) + set -- python /app/migrations/run.py "$@" + ;; + metrics) + set -- python -m litellm.proxy.prometheus_metrics_server "$@" + ;; + collector) + set -- python -m litellm.proxy.collector "$@" + ;; + *) + echo "docker-entrypoint.sh: unknown LITELLM_COMPONENT '$component'" >&2 + usage + exit 64 + ;; +esac + +if [ -n "${PROMETHEUS_MULTIPROC_DIR:-}" ]; then + case "$component" in + metrics|collector) mkdir -p "$PROMETHEUS_MULTIPROC_DIR" ;; + *) + mkdir -p "$PROMETHEUS_MULTIPROC_DIR" + rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db + ;; + esac +fi + +case "${USE_DDTRACE:-}" in + [Tt][Rr][Uu][Ee]) + export DD_TRACE_OPENAI_ENABLED="False" + exec ddtrace-run "$@" + ;; +esac + +exec "$@" diff --git a/docker/Dockerfile.database b/docker/Dockerfile.database deleted file mode 100644 index 61b6faae691..00000000000 --- a/docker/Dockerfile.database +++ /dev/null @@ -1,162 +0,0 @@ -# syntax=docker/dockerfile:1.7 - -# Base image for building -ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d - -# Runtime image -ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a -# Pinned by digest like the other base images; bump explicitly on Node upgrades. -ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43 -# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x) -ARG PGBOUNCER_VERSION=1.25.2 -ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332 - -FROM $UV_IMAGE AS uvbin - -FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder -ARG PGBOUNCER_VERSION -ARG PGBOUNCER_SHA256 -USER root -RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl -WORKDIR /build -RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \ - echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \ - tar xzf pgbouncer.tar.gz --strip-components=1 && \ - ./configure --prefix=/usr/local --with-openssl=/usr && \ - make -j"$(nproc)" pgbouncer && \ - install -m 0755 pgbouncer /usr/local/bin/pgbouncer - -# Admin UI builder. Pinned to the build platform so the architecture-independent -# Next.js static export compiles once natively even in a multi-arch build, -# instead of once per target arch under QEMU. -FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder - -ENV NEXT_TELEMETRY_DISABLED=1 \ - npm_config_fund=false \ - npm_config_audit=false - -WORKDIR /ui - -COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./ -RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline - -COPY ui/litellm-dashboard/ ./ -RUN npm run build - -FROM $LITELLM_BUILD_IMAGE AS builder - -WORKDIR /app -USER root - -COPY --from=uvbin /uv /usr/local/bin/uv -COPY --from=uvbin /uvx /usr/local/bin/uvx - -RUN apk add --no-cache \ - bash \ - gcc \ - python-3.13 \ - python-3.13-dev \ - openssl \ - openssl-dev \ - nodejs \ - npm \ - libsndfile - -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - UV_PYTHON_DOWNLOADS=0 \ - PATH="/app/.venv/bin:${PATH}" - -# Copy dependency metadata first for layer caching -COPY pyproject.toml uv.lock ./ -COPY enterprise/pyproject.toml enterprise/ -COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/ - -# Install third-party dependencies (cached unless pyproject.toml/uv.lock change) -RUN uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --extra bedrock-realtime \ - --python python3.13 - -# Copy full source tree -COPY . . - -# Replace the committed UI bundle with the one built from this exact source. -# Clearing first drops the committed bundle's content-hashed chunks that COPY -# would otherwise leave behind alongside the fresh ones. -RUN rm -rf litellm/proxy/_experimental/out -COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/ - -# Build Admin UI before final sync (applies the enterprise color override when present) -RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh - -# Install project and workspace packages (fast - deps already cached) -RUN uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --extra bedrock-realtime \ - --python python3.13 - -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma - -RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \ - sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh - -FROM $LITELLM_RUNTIME_IMAGE AS runtime - -USER root - -# node (without npm) is required by the prisma CLI at runtime -RUN apk add --no-cache bash openssl tzdata nodejs python-3.13 libsndfile libevent -COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer - -WORKDIR /app -ENV PATH="/app/.venv/bin:${PATH}" \ - PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \ - PRISMA_CLI_QUERY_ENGINE_TYPE=binary \ - PRISMA_OFFLINE_MODE=true - -# Copy only what runtime needs. The application is installed inside the venv; -# the rest of the builder's /app is source and build metadata that must not -# ship (manifest-scanning tools attribute everything in it to this image). -# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path. -COPY --from=builder /app/.venv /app/.venv -COPY --from=builder /app/docker /app/docker -COPY --from=builder /app/schema.prisma /app/schema.prisma -COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py -# enterprise/ is imported by source path at runtime (proxy_cli puts the -# working directory on sys.path; litellm/proxy/hooks resolves -# enterprise.enterprise_hooks from it) -COPY --from=builder /app/enterprise /app/enterprise -COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras -# Prisma CLI + engines are baked under /opt/prisma, a fixed path every -# runtime uid can read and that no cache volume mount shadows (unlike -# /app/.cache or $HOME/.cache in readOnlyRootFilesystem + emptyDir setups). -# The paths are pinned via PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH and -# recorded into the generated client at build time, so `prisma migrate -# deploy` on a fresh database needs no npm and no network access -# (#33650, #24554). -COPY --from=builder /opt/prisma /opt/prisma - -RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ - find /app/.venv -type d -path "*/tornado/test" -delete && \ - chmod -R a+rX /opt/prisma && \ - test -x /opt/prisma/binaries/node_modules/.bin/prisma && \ - test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \ - python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" - -EXPOSE 4000/tcp - -ENTRYPOINT ["docker/prod_entrypoint.sh"] -CMD ["--port", "4000"] diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root deleted file mode 100644 index eca12855afa..00000000000 --- a/docker/Dockerfile.non_root +++ /dev/null @@ -1,217 +0,0 @@ -# syntax=docker/dockerfile:1.7 - -# Base images -ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG PROXY_EXTRAS_SOURCE=published -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a -# Pinned by digest like the other base images; bump explicitly on Node upgrades. -ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43 -# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x) -ARG PGBOUNCER_VERSION=1.25.2 -ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332 - -FROM $UV_IMAGE AS uvbin - -FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder -ARG PGBOUNCER_VERSION -ARG PGBOUNCER_SHA256 -USER root -RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl -WORKDIR /build -RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \ - echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \ - tar xzf pgbouncer.tar.gz --strip-components=1 && \ - ./configure --prefix=/usr/local --with-openssl=/usr && \ - make -j"$(nproc)" pgbouncer && \ - install -m 0755 pgbouncer /usr/local/bin/pgbouncer - -# Admin UI builder. Pinned to the build platform so the architecture-independent -# Next.js static export compiles once natively even in a multi-arch build, -# instead of once per target arch under QEMU. -FROM --platform=$BUILDPLATFORM $UI_BUILD_IMAGE AS ui-builder - -ENV NEXT_TELEMETRY_DISABLED=1 \ - npm_config_fund=false \ - npm_config_audit=false - -WORKDIR /ui - -COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./ -RUN --mount=type=cache,target=/root/.npm npm ci --prefer-offline - -COPY ui/litellm-dashboard/ ./ -RUN npm run build - -FROM $LITELLM_BUILD_IMAGE AS builder -ARG PROXY_EXTRAS_SOURCE -WORKDIR /app -USER root - -COPY --from=uvbin /uv /usr/local/bin/uv -COPY --from=uvbin /uvx /usr/local/bin/uvx - -RUN for i in 1 2 3; do \ - apk add --no-cache \ - python-3.13 \ - python-3.13-dev \ - gcc \ - rust \ - bash \ - coreutils \ - curl \ - openssl \ - libsndfile \ - nodejs \ - npm && break || sleep 5; \ - done - -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - UV_PYTHON_DOWNLOADS=0 \ - PATH="/app/.venv/bin:${PATH}" \ - LITELLM_NON_ROOT=true \ - XDG_CACHE_HOME=/app/.cache - -# Copy dependency metadata first for layer caching -COPY pyproject.toml uv.lock ./ -COPY enterprise/pyproject.toml enterprise/ -COPY litellm-proxy-extras/pyproject.toml litellm-proxy-extras/ - -# Install third-party dependencies (cached unless pyproject.toml/uv.lock change) -RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \ - uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --extra bedrock-realtime \ - --python python3.13 - -# Copy full source tree -COPY . . - -# Replace the committed UI bundle with the one built from this exact source. -# Clearing first drops the committed bundle's content-hashed chunks that COPY -# would otherwise leave behind alongside the fresh ones. -RUN rm -rf litellm/proxy/_experimental/out -COPY --from=ui-builder /ui/out/. litellm/proxy/_experimental/out/ - -# Set non-root flag for build time consistency -ENV LITELLM_NON_ROOT=true - -RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \ - cp -r /app/litellm/proxy/_experimental/out/. /var/lib/litellm/ui/ && \ - cp /app/litellm/proxy/logo.png /var/lib/litellm/assets/logo.png && \ - touch /var/lib/litellm/ui/.litellm_ui_ready - -RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \ - if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \ - uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --extra bedrock-realtime \ - --python python3.13 \ - --no-sources-package litellm-proxy-extras; \ - else \ - uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra saml \ - --extra bedrock-realtime \ - --python python3.13; \ - fi - -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma - -RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \ - sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh - -FROM $LITELLM_RUNTIME_IMAGE AS runtime -ARG PROXY_EXTRAS_SOURCE -WORKDIR /app -USER root - -RUN for i in 1 2 3; do \ - apk upgrade --no-cache && break || sleep 5; \ - done && \ - for i in 1 2 3; do \ - apk add --no-cache python-3.13 bash openssl tzdata libsndfile nodejs libevent && break || sleep 5; \ - done -COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer - -# Copy only what runtime needs. The application is installed inside the venv; -# the rest of the builder's /app is source and build metadata that must not -# ship (manifest-scanning tools attribute everything in it to this image). -# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path. -COPY --from=builder /app/.venv /app/.venv -COPY --from=builder /app/docker /app/docker -COPY --from=builder /app/schema.prisma /app/schema.prisma -COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py -# enterprise/ is imported by source path at runtime (proxy_cli puts the -# working directory on sys.path; litellm/proxy/hooks resolves -# enterprise.enterprise_hooks from it) -COPY --from=builder /app/enterprise /app/enterprise -COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras -# Prisma CLI + engines are baked under /opt/prisma, a fixed path every runtime -# uid can read and that no cache volume mount shadows (unlike /app/.cache or -# $HOME/.cache under readOnlyRootFilesystem + emptyDir or arbitrary-uid setups). -# PRISMA_CLI_QUERY_ENGINE_TYPE=binary makes the CLI use the baked binary query -# engine directly, so `prisma migrate deploy` on a fresh database needs no npm -# and no network access; without it the CLI looks for the library engine, which -# prisma stopped baking, and falls back to a download that fails offline or as a -# non-writable uid (#33650, #24554). -COPY --from=builder /opt/prisma /opt/prisma -COPY --from=builder /var/lib/litellm/ui /var/lib/litellm/ui -COPY --from=builder /var/lib/litellm/assets /var/lib/litellm/assets - -# XDG_CACHE_HOME is intentionally left unset so it falls back to $HOME/.cache -# (/app/.cache, writable by the runtime uid). The prisma bake at the read-only -# /opt/prisma is anchored by PRISMA_BINARY_CACHE_DIR / PRISMA_CLI_PATH, so -# nothing needs XDG to point there; pointing it at the read-only bake would -# deny any XDG-aware library that writes a cache at runtime. -ENV PATH="/app/.venv/bin:${PATH}" \ - PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \ - PRISMA_CLI_QUERY_ENGINE_TYPE=binary \ - HOME=/app \ - LITELLM_NON_ROOT=true \ - PRISMA_SKIP_POSTINSTALL_GENERATE=1 \ - PRISMA_HIDE_UPDATE_MESSAGE=1 \ - PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \ - PRISMA_OFFLINE_MODE=true - -RUN mkdir -p /nonexistent /app/.cache /var/lib/litellm/assets /var/lib/litellm/ui && \ - chown -R nobody:nogroup /app /var/lib/litellm/ui /var/lib/litellm/assets /nonexistent && \ - PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \ - chown -R nobody:nogroup "$PRISMA_PATH" && \ - LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \ - [ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup "$LITELLM_PKG_MIGRATIONS_PATH" || true && \ - LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \ - chgrp -R 0 "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 "$LITELLM_PROXY_EXTRAS_PATH" || true && \ - chmod -R g=u "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u "$LITELLM_PROXY_EXTRAS_PATH" || true && \ - chmod -R g+w "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w "$LITELLM_PROXY_EXTRAS_PATH" || true && \ - chmod -R g+rX "$PRISMA_PATH" /var/lib/litellm/ui /var/lib/litellm/assets && \ - chmod -R a+rX /opt/prisma && \ - test -x /opt/prisma/binaries/node_modules/.bin/prisma && \ - test -f /opt/prisma/binaries/node_modules/prisma/build/index.js && \ - ls /opt/prisma/binaries/node_modules/@prisma/engines/query-engine-* >/dev/null 2>&1 && \ - python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" - -USER 65534 - -EXPOSE 4000/tcp - -ENTRYPOINT ["/app/docker/prod_entrypoint.sh"] -CMD ["--port", "4000"] diff --git a/docker/README.md b/docker/README.md index 376dc7b2d97..3c4907c0191 100644 --- a/docker/README.md +++ b/docker/README.md @@ -2,15 +2,15 @@ This guide provides instructions for building and running the LiteLLM application using Docker and Docker Compose. -> **Just want to run LiteLLM?** This guide builds from source. To run the published -> image instead, use `docker-compose.quickstart.yml` in this directory — the -> two-service stack (gateway + Postgres) that the -> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents: +> **Just want to run LiteLLM?** `docker-compose.yml` in the repository root runs +> the published image with a Postgres database, the stack the +> [Docker quickstart](https://docs.litellm.ai/docs/proxy/docker_quick_start) documents, +> and it works on its own outside a checkout: > > ```bash -> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml +> curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker-compose.yml > printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env -> docker compose -f docker-compose.quickstart.yml up -d +> docker compose up -d > ``` ## Prerequisites @@ -20,33 +20,52 @@ This guide provides instructions for building and running the LiteLLM applicatio ## Building and Running the Application -To build and run the application, you will use the `docker-compose.yml` file located in the root of the project. This file is configured to use the `Dockerfile.non_root` for a secure, non-root container environment. +The same `docker-compose.yml` builds from source when you pass `--build`: it builds the `Dockerfile` in the repository root, the one image LiteLLM ships. `docker-entrypoint.sh` next to it is the only entrypoint script; every container starts through it -### 1. Set the Master Key +## One image, many components -The application requires a `LITELLM_MASTER_KEY` for signing and validating tokens. You must set this key as an environment variable before running the application. +Every LiteLLM container runs the same image. The first word of the container command (or the `LITELLM_COMPONENT` environment variable when the command carries only flags) picks the process the container runs, and everything after it is handed to that process unchanged: -Create a `.env` file in the root of the project and add the following line: +| Component | Runs | Port | +|--------------|-----------------------------------------------------|------| +| `proxy` | `litellm ...` (everything in one process, default) | 4000 | +| `gateway` | `python -m gateway.launch ...` (inference routes) | 4000 | +| `backend` | `uvicorn backend.main:app ...` (management routes) | 4001 | +| `ui` | nginx serving the static admin UI | 3000 | +| `migrations` | `python migrations/run.py`, `prisma migrate deploy` then exit | | +| `metrics` | `python -m litellm.proxy.prometheus_metrics_server ...` | `--port` | +| `collector` | `python -m litellm.proxy.collector ...` | | -``` -LITELLM_MASTER_KEY=your-secret-key +```bash +docker run -p 4000:4000 litellm --config /app/config.yaml # proxy, exactly as before +docker run -p 4000:4000 litellm gateway --port 4000 # componentized data plane +docker run -p 4001:4001 -e LITELLM_COMPONENT=backend litellm # same, chosen through the env +docker run -p 3000:3000 --read-only --tmpfs /tmp litellm ui # admin UI behind nginx +docker run -e DATABASE_URL=... litellm migrations # one-off schema migration job +docker run -it litellm sh # anything else runs verbatim ``` -Replace `your-secret-key` with a strong, randomly generated secret. +PgBouncer is not a separate component: `LITELLM_PGBOUNCER_ENABLED=true` starts an in-container PgBouncer in front of `DATABASE_URL` inside `proxy` and `gateway`. `USE_DDTRACE=true` wraps whichever component runs with `ddtrace-run`, and `PROMETHEUS_MULTIPROC_DIR` is emptied of stale samples before any workers fork (the `metrics` and `collector` sidecars only read it, so their restart keeps the live samples) + +The image runs as uid `65532` (`nonroot` in the Wolfi base) and also works as an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, because everything it writes at runtime lives under `/app/.cache`, `/var/lib/litellm` and `/tmp`. Mount those (or set `readOnlyRootFilesystem` with emptyDirs there) for a read-only root filesystem. Prisma's CLI and engines are baked under `/opt/prisma`, so migrations need neither network nor a writable home + +### 1. Set the Master and Salt Keys + +The proxy signs virtual keys with `LITELLM_MASTER_KEY` and encrypts stored provider credentials with `LITELLM_SALT_KEY`. Compose reads both from a `.env` file in the directory you run it from, so generate them once: + +```bash +printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env +``` + +Keep the file: regenerating `LITELLM_SALT_KEY` makes credentials already stored in the database unreadable. Provider keys such as `OPENAI_API_KEY` go in the same file, and the whole file is passed to the container ### 2. Build and Run the Containers -Once you have set the `LITELLM_MASTER_KEY`, you can build and run the containers using the following command: - ```bash docker compose up -d --build ``` -This command will: - -- Build the Docker image using `Dockerfile.non_root`. -- Start the `litellm`, `litellm_db`, and `prometheus` services in detached mode (`-d`). -- The `--build` flag ensures that the image is rebuilt if there are any changes to the Dockerfile or the application code. +This command builds the image from the root `Dockerfile` and starts the `litellm` and `db` services in detached mode. Without `--build`, `docker compose up` pulls the published `main-stable` image instead. Add `--profile monitoring` to also start Prometheus on port 9090, scraping the proxy with the root `prometheus.yml` ### 3. Verifying the Application is Running @@ -70,34 +89,18 @@ To stop the running containers, use the following command: docker compose down ``` -## Hardened / Offline Testing +## Hardening -To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file: +The compose file runs the proxy the way a locked-down cluster would: as the image's non-root user with a read-only root filesystem, every capability dropped, `no-new-privileges` set, and tmpfs mounts only at `/tmp` and `/app/.cache`. The image is built for that, so a change that makes the proxy write anywhere else fails here before it fails in Kubernetes. To try an arbitrary uid in gid 0, the shape OpenShift `restricted-v2` assigns, add `user: "101:0"` to the `litellm` service + +Prisma's CLI and engines are baked under `/opt/prisma`, so migrations run without network access. Verify with: ```bash -docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache -docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d +docker run --rm --network none --entrypoint prisma docker.litellm.ai/berriai/litellm:main-stable --version ``` -This setup: -- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image. -- Runs the proxy as a non-root user with a read-only rootfs and only writable tmpfs mounts: - - `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`) - - `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`) -- Pre-builds and serves the admin UI from read-only paths: - - `/var/lib/litellm/ui` (pre-restructured Next.js UI with `.litellm_ui_ready` marker) - - `/var/lib/litellm/assets` (UI logos and assets) -- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines. - -You should also verify offline Prisma behaviour with: - -```bash -docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version -``` - -This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access. - ## Troubleshooting -- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project. -- **`Master key is not initialized`**: This error means the `LITELLM_MASTER_KEY` environment variable is not set. Make sure you have created a `.env` file in the project root with the `LITELLM_MASTER_KEY` defined. +- **`required variable LITELLM_MASTER_KEY is missing a value`**: Compose did not find a `.env` file with `LITELLM_MASTER_KEY` and `LITELLM_SALT_KEY` in the directory you ran it from. Generate one as shown above. +- **`password authentication failed for user "llmproxy"`**: the `litellm_postgres_data` volume was initialised by an older compose file with different credentials. `docker compose down -v` drops it and the next `up` recreates the database. +- **`build_admin_ui.sh: not found`**: the build context is wrong. Run `docker compose` from the root of the repository. diff --git a/docker/build_from_pip/Dockerfile.build_from_pip b/docker/build_from_pip/Dockerfile.build_from_pip deleted file mode 100644 index a5733f0e1a0..00000000000 --- a/docker/build_from_pip/Dockerfile.build_from_pip +++ /dev/null @@ -1,64 +0,0 @@ -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.10.9@sha256:10902f58a1606787602f303954cea099626a4adb02acbac4c69920fe9d278f82 -FROM $UV_IMAGE AS uvbin - -FROM python:3.13-slim@sha256:739e7213785e88c0f702dcdc12c0973afcbd606dbf021a589cab77d6b00b579d - -ARG LITELLM_VERSION=1.83.0 - -WORKDIR /app - -COPY --from=uvbin /uv /usr/local/bin/uv -COPY --from=uvbin /uvx /usr/local/bin/uvx - -RUN apt-get update && \ - apt-get install -y --no-install-recommends gcc libffi-dev nodejs npm && \ - rm -rf /var/lib/apt/lists/* - -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - PATH="/app/.venv/bin:${PATH}" - -COPY schema.prisma . - -# This image is specifically for validating/installing the published PyPI -# artifact, not the checked-out source tree. -# Keep the moved proxy-runtime packages explicit until the published PyPI -# artifact includes that extra; newer releases will simply dedupe these. -RUN uv venv --python python && \ - uv pip install --python /app/.venv/bin/python \ - "litellm[proxy,proxy-runtime]==${LITELLM_VERSION}" \ - "google-cloud-aiplatform==1.133.0" \ - "google-genai==1.37.0" \ - "anthropic[vertex]==0.84.0" \ - "grpcio==1.78.0" \ - "prometheus-client==0.20.0" \ - "langfuse==2.59.7" \ - "opentelemetry-api==1.28.0" \ - "opentelemetry-sdk==1.28.0" \ - "opentelemetry-exporter-otlp==1.28.0" \ - "ddtrace==4.11.0" \ - "sentry-sdk==2.21.0" \ - "mangum==0.17.0" \ - "azure-ai-contentsafety==1.0.0" \ - "azure-storage-file-datalake==12.20.0" \ - "pypdf==6.7.5" \ - "llm-sandbox==0.3.31" \ - "detect-secrets==1.5.0" \ - "prisma==0.11.0" \ - "openai==2.24.0" - -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma && \ - chmod -R a+rX /opt/prisma && \ - python -c "import sys; from prisma.client import BINARY_PATHS; bad = sorted(p for group in BINARY_PATHS.model_dump().values() for p in group.values() if not p.startswith('/opt/prisma/')); sys.exit('prisma engines baked outside /opt/prisma: %r' % bad) if bad else None" - -ENV PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries - -COPY docker/prod_entrypoint.sh /app/docker/prod_entrypoint.sh -RUN sed -i 's/\r$//' /app/docker/prod_entrypoint.sh && chmod +x /app/docker/prod_entrypoint.sh - -EXPOSE 4000/tcp - -ENTRYPOINT ["/app/docker/prod_entrypoint.sh"] -CMD ["--port", "4000"] diff --git a/docker/build_from_pip/Readme.md b/docker/build_from_pip/Readme.md deleted file mode 100644 index ad043588f33..00000000000 --- a/docker/build_from_pip/Readme.md +++ /dev/null @@ -1,9 +0,0 @@ -# Docker to build LiteLLM Proxy from litellm pip package - -### When to use this ? - -If you need to build LiteLLM Proxy from litellm pip package, you can use this Dockerfile as a reference. - -### Why build from pip package ? - -- If your company has a strict requirement around security / building images you can follow steps outlined here \ No newline at end of file diff --git a/docker/component_entrypoint.sh b/docker/component_entrypoint.sh deleted file mode 100755 index 173afafe1ad..00000000000 --- a/docker/component_entrypoint.sh +++ /dev/null @@ -1,16 +0,0 @@ -#!/bin/sh - -# stale samples from a previous container incarnation would be summed into the aggregate -if [ -n "$PROMETHEUS_MULTIPROC_DIR" ]; then - mkdir -p "$PROMETHEUS_MULTIPROC_DIR" - rm -f "$PROMETHEUS_MULTIPROC_DIR"/*.db -fi - -case "$USE_DDTRACE" in - [Tt][Rr][Uu][Ee]) - export DD_TRACE_OPENAI_ENABLED="False" - exec ddtrace-run "$@" - ;; -esac - -exec "$@" diff --git a/docker/docker-compose.quickstart.yml b/docker/docker-compose.quickstart.yml deleted file mode 100644 index 11631603a72..00000000000 --- a/docker/docker-compose.quickstart.yml +++ /dev/null @@ -1,41 +0,0 @@ -# LiteLLM quickstart stack: the gateway plus a Postgres database that stores -# models, virtual keys, and spend logs. Used by -# https://docs.litellm.ai/docs/proxy/docker_quick_start -# -# curl -sSLO https://github.com/BerriAI/litellm/raw/main/docker/docker-compose.quickstart.yml -# printf 'LITELLM_MASTER_KEY=sk-%s\nLITELLM_SALT_KEY=sk-%s\n' "$(openssl rand -hex 32)" "$(openssl rand -hex 32)" > .env -# docker compose -f docker-compose.quickstart.yml up -d -# -# Compose reads .env from this directory. Keep it: regenerating LITELLM_SALT_KEY -# makes credentials already stored in the database unreadable. For anything -# beyond local evaluation, pin the image to a specific release tag. -services: - litellm: - image: docker.litellm.ai/berriai/litellm:main-stable - ports: - - "4000:4000" - environment: - LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file} - LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file} - DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm - STORE_MODEL_IN_DB: "True" - depends_on: - db: - condition: service_healthy - - db: - image: postgres:16 - environment: - POSTGRES_USER: litellm - POSTGRES_PASSWORD: litellm - POSTGRES_DB: litellm - healthcheck: - test: ["CMD-SHELL", "pg_isready -U litellm"] - interval: 5s - timeout: 5s - retries: 10 - volumes: - - postgres_data:/var/lib/postgresql/data - -volumes: - postgres_data: diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh deleted file mode 100755 index 003d9b21db8..00000000000 --- a/docker/entrypoint.sh +++ /dev/null @@ -1,16 +0,0 @@ -#!/bin/bash -set -euo pipefail - -REPO_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" -VENV_PYTHON="$REPO_ROOT/.venv/bin/python" -MIGRATION_SCRIPT="$REPO_ROOT/litellm/proxy/prisma_migration.py" - -if [ -x "$VENV_PYTHON" ]; then - "$VENV_PYTHON" "$MIGRATION_SCRIPT" -elif command -v uv >/dev/null 2>&1; then - (cd "$REPO_ROOT" && uv run --no-sync python "$MIGRATION_SCRIPT") -else - python3 "$MIGRATION_SCRIPT" -fi - -echo "Migration script ran successfully!" diff --git a/docker/install_auto_router.sh b/docker/install_auto_router.sh deleted file mode 100755 index 4fedf201b41..00000000000 --- a/docker/install_auto_router.sh +++ /dev/null @@ -1,4 +0,0 @@ -#!/bin/bash -set -euo pipefail - -# semantic-router dependencies are installed via `uv sync`. diff --git a/docker/prod_entrypoint.sh b/docker/prod_entrypoint.sh deleted file mode 100644 index 630eb6b065b..00000000000 --- a/docker/prod_entrypoint.sh +++ /dev/null @@ -1,10 +0,0 @@ -#!/bin/sh - -case "$USE_DDTRACE" in - [Tt][Rr][Uu][Ee]) - export DD_TRACE_OPENAI_ENABLED="False" - exec ddtrace-run litellm "$@" - ;; -esac - -exec litellm "$@" diff --git a/docker/tests/nonroot.yaml b/docker/tests/nonroot.yaml deleted file mode 100644 index 36118ca8c59..00000000000 --- a/docker/tests/nonroot.yaml +++ /dev/null @@ -1,18 +0,0 @@ -schemaVersion: 2.0.0 - -metadataTest: - entrypoint: ["docker/prod_entrypoint.sh"] - user: "65534" - workdir: "/app" - -fileExistenceTests: - - name: "Prisma Folder" - path: "/usr/local/lib/python3.13/site-packages/prisma/" - shouldExist: true - uid: 65534 - gid: 65534 - - name: "Prisma Schema" - path: "/usr/local/lib/python3.13/site-packages/prisma/schema.prisma" - shouldExist: true - uid: 65534 - gid: 65534 diff --git a/gateway/Dockerfile b/gateway/Dockerfile deleted file mode 100644 index 8045a8b64cb..00000000000 --- a/gateway/Dockerfile +++ /dev/null @@ -1,126 +0,0 @@ -ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a -# Checksum from https://www.pgbouncer.org/downloads/ (the Wolfi repo only carries 1.24.x) -ARG PGBOUNCER_VERSION=1.25.2 -ARG PGBOUNCER_SHA256=924ad35113fd0a71c8e2dbe85b5d03445532e2b7b37a9f8a48983beea238b332 - -FROM $UV_IMAGE AS uvbin - -FROM $LITELLM_BUILD_IMAGE AS pgbouncer-builder -ARG PGBOUNCER_VERSION -ARG PGBOUNCER_SHA256 -USER root -RUN apk add --no-cache build-base pkgconf libevent-dev openssl-dev curl -WORKDIR /build -RUN curl -fsSL -o pgbouncer.tar.gz "https://www.pgbouncer.org/downloads/files/${PGBOUNCER_VERSION}/pgbouncer-${PGBOUNCER_VERSION}.tar.gz" && \ - echo "${PGBOUNCER_SHA256} pgbouncer.tar.gz" | sha256sum -c - && \ - tar xzf pgbouncer.tar.gz --strip-components=1 && \ - ./configure --prefix=/usr/local --with-openssl=/usr && \ - make -j"$(nproc)" pgbouncer && \ - install -m 0755 pgbouncer /usr/local/bin/pgbouncer - -# ---------- Builder ---------- -FROM $LITELLM_BUILD_IMAGE AS builder - -WORKDIR /app -USER root - -COPY --from=uvbin /uv /uvx /usr/local/bin/ - -# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE -# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi -# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes. -RUN for i in 1 2 3; do \ - apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -# UV_COMPILE_BYTECODE=1 precompiles .pyc at install time → faster cold start. -# UV_LINK_MODE=copy avoids hardlink warnings when uv installs from a -# BuildKit cache mount (different filesystem). -# UV_PYTHON_DOWNLOADS=0 force uv to use the apk-installed CPython instead of -# silently pulling a managed interpreter. -# PRISMA_USE_GLOBAL_NODE explicit (matches default) so an env override can't -# silently re-enable nodeenv's Node download. -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - UV_COMPILE_BYTECODE=1 \ - UV_PYTHON_DOWNLOADS=0 \ - PRISMA_USE_GLOBAL_NODE=true \ - PATH="/app/.venv/bin:${PATH}" - -# Stage 1 — install dependencies only. -RUN --mount=type=cache,target=/root/.cache/uv \ - --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ - --mount=type=bind,source=uv.lock,target=uv.lock \ - --mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \ - --mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \ - uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra bedrock-realtime \ - --python python3.13 - -# Stage 2 — copy source and install the project + workspace members. -COPY . . - -RUN --mount=type=cache,target=/root/.cache/uv \ - uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra proxy-runtime \ - --extra extra_proxy \ - --extra semantic-router \ - --extra bedrock-realtime \ - --python python3.13 - -# PYTHONPATH=/app makes the source tree shadow the installed package, so the -# compiled Rust extension must live next to the source or it is never imported. -RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/ - -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma - -RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component_entrypoint.sh - -# ---------- Runtime ---------- -FROM $LITELLM_RUNTIME_IMAGE AS runtime - -USER root - -RUN for i in 1 2 3; do \ - apk add --no-cache bash openssl tzdata python-3.13 libsndfile libatomic libevent && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532) with -# /home/nonroot. We run the proxy as that user. -WORKDIR /app -ENV HOME=/home/nonroot \ - PATH="/app/.venv/bin:${PATH}" \ - PYTHONPATH="/app" \ - PYTHONDONTWRITEBYTECODE=1 \ - PYTHONUNBUFFERED=1 \ - PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries - -COPY --from=builder --chown=nonroot:nonroot /app /app -COPY --from=builder /opt/prisma /opt/prisma -COPY --from=pgbouncer-builder /usr/local/bin/pgbouncer /usr/local/bin/pgbouncer - -RUN find /app/.venv -type f -path "*/tornado/test/*" -delete && \ - find /app/.venv -type d -path "*/tornado/test" -delete && \ - chmod -R a+rX /opt/prisma && \ - python -c "from prisma.client import BINARY_PATHS; paths = list(BINARY_PATHS.query_engine.values()); assert paths and all(p.startswith('/opt/prisma/') for p in paths), paths" && \ - python -c "import litellm; from litellm.rust_bridge.loader import native_bridge_available; assert litellm.__file__ == '/app/litellm/__init__.py', litellm.__file__; assert native_bridge_available()" - -USER nonroot - -EXPOSE 4000/tcp - -ENTRYPOINT ["sh", "-c", "exec /app/docker/component_entrypoint.sh python -m gateway.launch --workers \"${NUM_WORKERS:-1}\" \"$@\"", "--"] -CMD ["--host", "0.0.0.0", "--port", "4000"] diff --git a/litellm/proxy/start.sh b/litellm/proxy/start.sh deleted file mode 100755 index 44df50aaabf..00000000000 --- a/litellm/proxy/start.sh +++ /dev/null @@ -1,2 +0,0 @@ -#!/bin/bash -python3 proxy_cli.py \ No newline at end of file diff --git a/migrations/Dockerfile b/migrations/Dockerfile deleted file mode 100644 index 255b94b0ea8..00000000000 --- a/migrations/Dockerfile +++ /dev/null @@ -1,120 +0,0 @@ -ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d -ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a - -FROM $UV_IMAGE AS uvbin - -# ---------- Builder ---------- -# -# Minimal install for `prisma migrate deploy`. We deliberately skip the heavy -# `proxy-runtime` (otel, sentry, ddtrace, pypdf, google-genai, anthropic-vertex, -# ...) and `semantic-router` extras that the gateway/backend pull in — the -# migration engine doesn't need them. We DO install `--extra proxy` so the -# DB-URL helper from `litellm.proxy.auth.rds_iam_token` is importable, which -# is how the gateway and backend assemble `DATABASE_URL` at pod startup when -# `IAM_TOKEN_DB_AUTH=true` (see backend/main.py:17, gateway/main.py:22). And -# `--extra extra_proxy` provides the `prisma` CLI + the secret-manager -# backends `litellm.secret_managers.main` lazily imports. -# -# `prisma generate` runs once at BUILD time to (a) install the Node-based -# Prisma CLI into the binary cache and (b) download the migration / query -# engine binaries. The Python client it also produces is unused by this -# image's runtime entrypoint — that's fine, it's a few hundred KB and the -# alternative (`prisma py fetch`) doesn't reliably trigger engine downloads -# under nodeenv. Crucially we do NOT run `prisma generate` at RUNTIME; the -# old migration job did, on every pod start, which is the wasteful behaviour -# the componentization is fixing. -FROM $LITELLM_BUILD_IMAGE AS builder - -WORKDIR /app -USER root - -COPY --from=uvbin /uv /uvx /usr/local/bin/ - -# nodejs/npm so `prisma generate` uses Wolfi's Node via PRISMA_USE_GLOBAL_NODE -# instead of nodeenv downloading one whose dynamic deps may not be in Wolfi -# (e.g. Node 26.2.0 needs libatomic). Retry for transient apk.cgr.dev flakes. -RUN for i in 1 2 3; do \ - apk add --no-cache bash gcc python-3.13 python-3.13-dev openssl openssl-dev libsndfile nodejs npm && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -ENV UV_PROJECT_ENVIRONMENT=/app/.venv \ - UV_LINK_MODE=copy \ - UV_COMPILE_BYTECODE=1 \ - UV_PYTHON_DOWNLOADS=0 \ - PRISMA_USE_GLOBAL_NODE=true \ - PATH="/app/.venv/bin:${PATH}" - -# Stage 1 — install third-party deps only (cached by pyproject.toml/uv.lock). -RUN --mount=type=cache,target=/root/.cache/uv \ - --mount=type=bind,source=pyproject.toml,target=pyproject.toml \ - --mount=type=bind,source=uv.lock,target=uv.lock \ - --mount=type=bind,source=enterprise/pyproject.toml,target=enterprise/pyproject.toml \ - --mount=type=bind,source=litellm-proxy-extras/pyproject.toml,target=litellm-proxy-extras/pyproject.toml \ - uv sync --frozen --no-install-project --no-install-workspace --no-default-groups --no-editable \ - --extra proxy \ - --extra extra_proxy \ - --python python3.13 - -# Stage 2 — copy source and install the project + workspace members. -COPY . . - -RUN --mount=type=cache,target=/root/.cache/uv \ - uv sync --frozen --no-default-groups --no-editable \ - --extra proxy \ - --extra extra_proxy \ - --python python3.13 - -RUN cp "$(python -c 'import sysconfig; print(sysconfig.get_paths()["purelib"])')"/litellm/rust_bridge/_native*.so litellm/rust_bridge/ - -COPY migrations/run.py /app/run.py - -# Pre-warm the Prisma binary cache so the Job pod doesn't reach the -# internet on first start. This matches what the backend Dockerfile does: -# `prisma generate` runs nodeenv (downloads Node), installs the prisma npm -# CLI, downloads the engine binaries for each `binaryTarget` in -# schema.prisma, AND emits the generated Python client. We don't need the -# client at runtime — the migration job invokes `prisma migrate deploy` -# via subprocess — but having it cached is harmless and the alternative -# (`prisma py fetch`) doesn't reliably trigger engine downloads. -RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - npm_config_cache=/root/.npm \ - prisma generate --schema=./schema.prisma - -# ---------- Runtime ---------- -FROM $LITELLM_RUNTIME_IMAGE AS runtime - -USER root - -RUN for i in 1 2 3; do \ - apk add --no-cache bash openssl tzdata python-3.13 nodejs libsndfile libatomic && break; \ - [ $i = 3 ] && { echo "apk add failed after 3 retries" >&2; exit 1; }; \ - sleep 5; \ - done - -# wolfi-base ships an unprivileged `nonroot` account (UID/GID 65532). The -# Prisma engine binaries are dynamically linked against libssl/libcrypto, so -# openssl stays in the runtime layer. -WORKDIR /app -ENV HOME=/home/nonroot \ - PATH="/app/.venv/bin:${PATH}" \ - PYTHONPATH="/app" \ - PYTHONDONTWRITEBYTECODE=1 \ - PYTHONUNBUFFERED=1 \ - PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \ - PRISMA_CLI_PATH=/opt/prisma/binaries/node_modules/.bin/prisma \ - PRISMA_CLI_QUERY_ENGINE_TYPE=binary \ - PRISMA_OFFLINE_MODE=true - -COPY --from=builder --chown=nonroot:nonroot /app /app -COPY --from=builder /opt/prisma /opt/prisma - -RUN chmod -R a+rX /opt/prisma && \ - test -x /opt/prisma/binaries/node_modules/.bin/prisma && \ - test -f /opt/prisma/binaries/node_modules/prisma/build/index.js - -USER nonroot - -ENTRYPOINT ["python3", "/app/run.py"] diff --git a/docker/build_from_pip/litellm_config.yaml b/tests/basic_proxy_startup_tests/build_from_pip_config.yaml similarity index 77% rename from docker/build_from_pip/litellm_config.yaml rename to tests/basic_proxy_startup_tests/build_from_pip_config.yaml index f54647853ef..ff0db25e33f 100644 --- a/docker/build_from_pip/litellm_config.yaml +++ b/tests/basic_proxy_startup_tests/build_from_pip_config.yaml @@ -5,5 +5,5 @@ model_list: api_key: fake-key api_base: os.environ/FAKE_OPENAI_API_BASE -general_settings: - alerting: ["slack"] \ No newline at end of file +general_settings: + alerting: ["slack"] diff --git a/tests/e2e/claude_code/cron_vm/Dockerfile b/tests/e2e/claude_code/cron_vm/Dockerfile deleted file mode 100644 index 05c1387c5a2..00000000000 --- a/tests/e2e/claude_code/cron_vm/Dockerfile +++ /dev/null @@ -1,34 +0,0 @@ -FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251 - -ARG GH_VERSION=2.101.0 -ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8 -ARG UV_VERSION=0.10.9 -ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594 - -SHELL ["/bin/bash", "-o", "pipefail", "-c"] - -RUN apt-get update \ - && apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \ - && rm -rf /var/lib/apt/lists/* - -RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \ - && echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \ - && tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \ - && rm /tmp/gh.tar.gz - -RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \ - && echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \ - && tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \ - && rm /tmp/uv.tar.gz - -RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator - -ENV HOME=/home/populator \ - LITELLM_REPO=/opt/litellm \ - DISABLE_AUTOUPDATER=1 - -COPY --chown=populator:populator . /opt/litellm/tests/e2e/ - -USER populator -WORKDIR /home/populator -CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"] diff --git a/tests/e2e/claude_code/cron_vm/README.md b/tests/e2e/claude_code/cron_vm/README.md index 4c16b6e237a..ee9f7a2d5f1 100644 --- a/tests/e2e/claude_code/cron_vm/README.md +++ b/tests/e2e/claude_code/cron_vm/README.md @@ -1,12 +1,12 @@ # Render cron job for the Claude Code compatibility-matrix populator The populator runs daily as the Render cron job `litellm-compat-matrix` -(Docker runtime, built from the `Dockerfile` in this directory) rather +(Docker runtime; the image recipe is in [Image](#image) below) rather than as a GitHub Action or on a dedicated VM. Trade-offs: -- ✅ No machine to keep on or patch. Render builds the image from this - directory on every push to `main` that touches `tests/e2e/**` and - runs it on the schedule. +- ✅ No machine to keep on or patch. Render builds the image on every + push to `main` that touches `tests/e2e/**` and runs it on the + schedule. - ✅ Credentials live in Render env vars and secret files, scoped to this one service, instead of on a VM filesystem. - ✅ The publish token still uses the `mateo-berri` account, which is a @@ -25,7 +25,6 @@ than as a GitHub Action or on a dedicated VM. Trade-offs: | File | Purpose | | --- | --- | -| `Dockerfile` | The image Render builds: Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the non-root user `populator` (uid/gid 1000, which is what Render's secret files are readable by). | | `run_daily.sh` | The actual cron job. Resolves versions, clones the worktree, installs the Claude Code CLI under test, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. | | `install_claude_code.sh` | Downloads one Claude Code release (` `) from the vendor's native release channel, verifies it against the sha256 in that release's `manifest.json`, and refuses a binary whose `--version` disagrees. Run by the cron and by the `compat-matrix-image` GitHub workflow. | | `build_matrix.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.build_from_paths`. Exists only because the bash script needs *some* way to render the per-cell aggregation, and the builder is already Python. | @@ -106,7 +105,7 @@ the same values if it ever has to be rebuilt. | Workspace | Litellm (the one that already builds the other litellm services) | | Type | Cron job, Docker runtime | | Repo / branch | `BerriAI/litellm` @ `main` | -| Dockerfile path | `tests/e2e/claude_code/cron_vm/Dockerfile` | +| Image | Built from the recipe under [Image](#image). Render only builds a Dockerfile that lives in the connected repo, and this repo ships exactly one Dockerfile (the LiteLLM image in its root), so the service has to point at a copy of the recipe kept with the service or at a prebuilt image | | Docker build context | `tests/e2e` (the repo root `.dockerignore` excludes `tests`, so the context has to start below it) | | Build filter | included paths `tests/e2e/**` | | Schedule | `0 6 * * *` (06:00 UTC daily) | @@ -117,7 +116,7 @@ the same values if it ever has to be rebuilt. Render mounts secret files at `/etc/secrets/`, which is where `CREDENTIALS_DIRECTORY` and `GOOGLE_APPLICATION_CREDENTIALS` in the env example point. Render also passes env vars to `docker build` as build -args, which is why the `Dockerfile` declares no `ARG` that could ever +args, which is why the image recipe declares no `ARG` that could ever be given a secret's name. Creating it through the API looks like this (fill `envVars` and @@ -145,13 +144,59 @@ curl -fsS https://api.render.com/v1/services \ "plan": "4c-16g", "region": "oregon", "envSpecificDetails": { - "dockerfilePath": "tests/e2e/claude_code/cron_vm/Dockerfile", + "dockerfilePath": "", "dockerContext": "tests/e2e" } } }' ``` +## Image + +Debian bookworm-slim plus pinned, checksum-verified `gh` and `uv`, with +this `tests/e2e/` tree copied to `/opt/litellm/tests/e2e/`. Runs as the +non-root user `populator` (uid/gid 1000, which is what Render's secret +files are readable by). This is the recipe the Render service builds; +it lives with the service because the repo ships exactly one Dockerfile, +the LiteLLM image in its root + +```dockerfile +FROM debian:bookworm-slim@sha256:3783cc01769c7b2b1b83a5c5ad96c815348e28ed7da68e2e3687004faa906251 + +ARG GH_VERSION=2.101.0 +ARG GH_SHA256=9bca2d1c16825f109907a23307628a2f0698fbf99662b73a5cf0b020293072b8 +ARG UV_VERSION=0.10.9 +ARG UV_SHA256=20d79708222611fa540b5c9ed84f352bcd3937740e51aacc0f8b15b271c57594 + +SHELL ["/bin/bash", "-o", "pipefail", "-c"] + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ca-certificates curl git jq procps iproute2 \ + && rm -rf /var/lib/apt/lists/* + +RUN curl -fsSLo /tmp/gh.tar.gz "https://github.com/cli/cli/releases/download/v${GH_VERSION}/gh_${GH_VERSION}_linux_amd64.tar.gz" \ + && echo "${GH_SHA256} /tmp/gh.tar.gz" | sha256sum -c - \ + && tar -xzf /tmp/gh.tar.gz -C /usr/local/bin --strip-components=2 "gh_${GH_VERSION}_linux_amd64/bin/gh" \ + && rm /tmp/gh.tar.gz + +RUN curl -fsSLo /tmp/uv.tar.gz "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-x86_64-unknown-linux-gnu.tar.gz" \ + && echo "${UV_SHA256} /tmp/uv.tar.gz" | sha256sum -c - \ + && tar -xzf /tmp/uv.tar.gz -C /usr/local/bin --strip-components=1 uv-x86_64-unknown-linux-gnu/uv \ + && rm /tmp/uv.tar.gz + +RUN groupadd --gid 1000 populator && useradd --uid 1000 --gid 1000 --create-home populator + +ENV HOME=/home/populator \ + LITELLM_REPO=/opt/litellm \ + DISABLE_AUTOUPDATER=1 + +COPY --chown=populator:populator . /opt/litellm/tests/e2e/ + +USER populator +WORKDIR /home/populator +CMD ["/opt/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh"] +``` + ## Operating it ```bash @@ -185,7 +230,7 @@ curl -fsS "https://api.render.com/v1/services/${CRON_ID}/deploys?limit=1" \ # Build and run the image locally (docker on Apple silicon needs the # platform flag; the context is tests/e2e, see the table above). docker build --platform linux/amd64 \ - -f tests/e2e/claude_code/cron_vm/Dockerfile -t compat-matrix tests/e2e + -f /path/to/compat-matrix.Dockerfile -t compat-matrix tests/e2e docker run --rm --platform linux/amd64 \ --env-file litellm-compat-matrix.env -e SKIP_PUBLISH=1 \ -v "$PWD/secrets:/etc/secrets:ro" compat-matrix @@ -234,8 +279,8 @@ docker run --rm --platform linux/amd64 \ A CLI release that breaks a cell shows up as a green→red flip, which withholds auto-merge on that day's docs PR for review. To rerun the matrix on one specific CLI, set `CLAUDE_CODE_VERSION` on the run. - `gh` and `uv` stay pinned in the `Dockerfile`; bump them in a PR with - the checksum from the release's `gh__checksums.txt` and the + `gh` and `uv` stay pinned in the image recipe; bump them with the + checksum from the release's `gh__checksums.txt` and the tarball's `.sha256` sidecar respectively. - **A local build on Apple silicon only proves the image assembles.** Under QEMU the Claude Code binary (a Bun executable) dies with diff --git a/tests/e2e/claude_code/cron_vm/run_daily.sh b/tests/e2e/claude_code/cron_vm/run_daily.sh index af01057eff7..35c3d244f81 100755 --- a/tests/e2e/claude_code/cron_vm/run_daily.sh +++ b/tests/e2e/claude_code/cron_vm/run_daily.sh @@ -1,8 +1,8 @@ #!/usr/bin/env bash # Daily Claude Code compatibility-matrix populator. # -# Runs daily as the Render cron job `litellm-compat-matrix`, built from -# the Dockerfile in this directory (see README.md). The flow is: +# Runs daily as the Render cron job `litellm-compat-matrix` (see +# README.md for the image it runs in). The flow is: # # 1. Resolve the latest LiteLLM final release tag from the GitHub # Releases API. diff --git a/tests/proxy_migration_tests/test_component_image_serves_offline.py b/tests/proxy_migration_tests/test_component_image_serves_offline.py index 528c2960072..d19a7c7139f 100644 --- a/tests/proxy_migration_tests/test_component_image_serves_offline.py +++ b/tests/proxy_migration_tests/test_component_image_serves_offline.py @@ -1,4 +1,4 @@ -"""Image-level regression net for the prisma bake in the componentized images. +"""Image-level regression net for the prisma bake in the `gateway` and `backend` components. The gateway and backend serve requests; they never shell out to the Prisma CLI (``PrismaManager.setup_database`` is reachable only from ``proxy_cli.py``, which @@ -35,7 +35,8 @@ import pytest IMAGE = os.getenv("LITELLM_IMAGE") POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine") CURL_IMAGE = os.getenv("LITELLM_TEST_CURL_IMAGE", "curlimages/curl:8.11.1") -COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", "4000") +COMPONENT = os.getenv("LITELLM_COMPONENT", "gateway") +COMPONENT_PORT = os.getenv("LITELLM_COMPONENT_PORT", {"gateway": "4000", "backend": "4001"}[COMPONENT]) NON_ROOT_UID = "12345:0" STARTUP_TIMEOUT_SECONDS = int(os.getenv("LITELLM_COMPONENT_STARTUP_TIMEOUT", "180")) @@ -63,7 +64,7 @@ def offline_stack(): The container runs with DISABLE_SCHEMA_UPDATE, since applying the schema is the migration job's responsibility in this topology and needs the Prisma CLI - these images deliberately omit, and with LITELLM_LOCAL_MODEL_COST_MAP, or the + these components never invoke, and with LITELLM_LOCAL_MODEL_COST_MAP, or the proxy spends the whole startup budget timing out on a cost-map fetch over the network it does not have. @@ -92,7 +93,7 @@ def offline_stack(): "-e", "LITELLM_MASTER_KEY=sk-component-serve-test", "-e", "DISABLE_SCHEMA_UPDATE=true", "-e", "LITELLM_LOCAL_MODEL_COST_MAP=True", - IMAGE, + IMAGE, COMPONENT, ) yield network, component finally: diff --git a/tests/proxy_migration_tests/test_offline_image_migration.py b/tests/proxy_migration_tests/test_offline_image_migration.py index be2f0537576..2463fd0fd99 100644 --- a/tests/proxy_migration_tests/test_offline_image_migration.py +++ b/tests/proxy_migration_tests/test_offline_image_migration.py @@ -1,6 +1,6 @@ """Image-level regression net for the prisma bake in the shipped runtime image. -Boots a built image's migration entrypoint the way an OpenShift / air-gapped +Boots a built image's `migrations` component the way an OpenShift / air-gapped deployment does (an internal-only network with no egress, an arbitrary non-root uid in GID 0) against a brand-new Postgres, and asserts the schema was created. @@ -14,6 +14,7 @@ the normal unit-test run and exercised only where an image has been built (the image-scan workflow). Requires a working docker CLI. """ +import shlex import shutil import subprocess import uuid @@ -26,10 +27,7 @@ POSTGRES_IMAGE = os.getenv("LITELLM_TEST_POSTGRES_IMAGE", "postgres:16-alpine") MIN_TABLES = int(os.getenv("LITELLM_TEST_MIN_TABLES", "20")) NON_ROOT_UID = "12345:0" # arbitrary uid in GID 0, as OpenShift restricted-v2 assigns -MIGRATION_INTERPRETER = os.getenv("LITELLM_MIGRATION_INTERPRETER", "python") -MIGRATION_SCRIPT = os.getenv( - "LITELLM_MIGRATION_SCRIPT", "litellm/proxy/prisma_migration.py" -) +MIGRATION_ARGS = tuple(shlex.split(os.getenv("LITELLM_MIGRATION_ARGS", "migrations"))) pytestmark = [ pytest.mark.skipif(IMAGE is None, reason="requires a built image (set LITELLM_IMAGE)"), @@ -113,8 +111,7 @@ def test_migration_offline_as_non_root_uid(offline_postgres): "-e", f"DATABASE_URL=postgresql://postgres:pw@{pg}:5432/litellm", "-e", "LITELLM_MASTER_KEY=sk-offline-migration-test", "-e", "DISABLE_SCHEMA_UPDATE=false", - "-w", "/app", "--entrypoint", MIGRATION_INTERPRETER, - IMAGE, MIGRATION_SCRIPT, + IMAGE, *MIGRATION_ARGS, check=False, ) tables = _table_count(pg) diff --git a/tests/proxy_migration_tests/test_ui_image_serves_offline.py b/tests/proxy_migration_tests/test_ui_image_serves_offline.py index 5ff68effd7f..b3092887b75 100644 --- a/tests/proxy_migration_tests/test_ui_image_serves_offline.py +++ b/tests/proxy_migration_tests/test_ui_image_serves_offline.py @@ -1,7 +1,7 @@ -"""Image-level regression net for arbitrary-uid boot of the UI image. +"""Image-level regression net for arbitrary-uid boot of the `ui` component. OpenShift ``restricted-v2`` ignores the image ``USER`` and assigns an -arbitrary uid in GID 0. The stock nginx base expects to start as root, so +arbitrary uid in GID 0. A stock nginx install expects to start as root, so its cache (``/var/cache/nginx``) and pid (``/run``) paths are root-owned 755 and the master process dies at startup with ``mkdir() "/var/cache/nginx/client_temp" failed (13: Permission denied)``. @@ -63,7 +63,7 @@ def ui_container() -> Iterator[tuple[str, str]]: "run", "-d", "--name", container, "--network", network, "--user", ARBITRARY_UID, "--read-only", "--tmpfs", "/tmp", - IMAGE, + IMAGE, "ui", ) yield network, container finally: diff --git a/tests/unit/test_component_entrypoint.py b/tests/unit/test_docker_entrypoint.py similarity index 52% rename from tests/unit/test_component_entrypoint.py rename to tests/unit/test_docker_entrypoint.py index 0c2a533b8bc..7b8225b8b54 100644 --- a/tests/unit/test_component_entrypoint.py +++ b/tests/unit/test_docker_entrypoint.py @@ -1,5 +1,5 @@ -"""Unit tests for `docker/component_entrypoint.sh` and its wiring into the -componentized `gateway` / `backend` images and Terraform deployments.""" +"""Unit tests for `docker-entrypoint.sh`, the component dispatcher every shipped +container runs, and for the Terraform launch commands that must agree with it.""" import json import os @@ -11,15 +11,13 @@ from pathlib import Path import pytest REPO_ROOT = Path(__file__).resolve().parents[2] -COMPONENT_ENTRYPOINT = REPO_ROOT / "docker" / "component_entrypoint.sh" -PROD_ENTRYPOINT = REPO_ROOT / "docker" / "prod_entrypoint.sh" -GATEWAY_DOCKERFILE = REPO_ROOT / "gateway" / "Dockerfile" -BACKEND_DOCKERFILE = REPO_ROOT / "backend" / "Dockerfile" -BUILD_FROM_PIP_DOCKERFILE = REPO_ROOT / "docker" / "build_from_pip" / "Dockerfile.build_from_pip" +DOCKER_ENTRYPOINT = REPO_ROOT / "docker-entrypoint.sh" +DOCKERFILE = REPO_ROOT / "Dockerfile" TERRAFORM_ECS = REPO_ROOT / "terraform" / "litellm" / "aws" / "ecs.tf" TERRAFORM_CLOUDRUN = REPO_ROOT / "terraform" / "litellm" / "gcp" / "cloudrun.tf" -IMAGE_ENTRYPOINT_PATH = "/app/docker/component_entrypoint.sh" +IMAGE_ENTRYPOINT_PATH = "/app/docker-entrypoint.sh" +STUBBED_EXECUTABLES = ("ddtrace-run", "uvicorn", "python", "litellm", "nginx") TRUTHY_USE_DDTRACE = ("true", "True", "TRUE", "tRuE") FALSY_USE_DDTRACE = (None, "", "false", "False", "1", "yes", "on", "truex") @@ -37,7 +35,6 @@ _STUB_TEMPLATE = """#!/bin/sh _ENTRYPOINT_RE = re.compile(r"^ENTRYPOINT\s+(\[.*\])\s*$", re.MULTILINE) _CMD_RE = re.compile(r"^CMD\s+(\[.*\])\s*$", re.MULTILINE) -_COPY_RE = re.compile(r"^COPY\s+(?!--from)(\S+)\s+(\S+)\s*$", re.MULTILINE) _APP_TARGET_RE = re.compile(r"(?:gateway|backend)\.main:app|gateway\.launch") _TF_STRING_LOCAL_RE = re.compile(r'^\s*(\w+)\s*=\s*"((?:[^"\\]|\\.)*)"\s*$', re.MULTILINE) _TF_INTERPOLATION_RE = re.compile(r"\$\{(local|var)\.(\w+)\}") @@ -58,53 +55,39 @@ def _write_stubs(bin_dir: Path, names: tuple[str, ...]) -> None: stub.chmod(0o755) -def _run_entrypoint( - script: Path, - argv: tuple[str, ...], - use_ddtrace: str | None, - tmp_path: Path, -) -> tuple[str, ...]: - """Run `script` with stubbed executables on PATH and return the recorded lines.""" - bin_dir = tmp_path / "bin" - bin_dir.mkdir(parents=True) - _write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python", "litellm")) - record = tmp_path / "record.txt" - - env = { - **os.environ, +def _entrypoint_env(bin_dir: Path, record: Path, overrides: dict[str, str | None]) -> dict[str, str]: + """The container-like environment the entrypoint runs under, with `overrides` applied (None unsets).""" + cleared = ("USE_DDTRACE", "DD_TRACE_OPENAI_ENABLED", "LITELLM_COMPONENT", "NUM_WORKERS", *overrides) + return { + **{k: v for k, v in os.environ.items() if k not in cleared}, + **{k: v for k, v in overrides.items() if v is not None}, "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", "RECORD": str(record), "PYTHONPATH": PYTHONPATH_SENTINEL, } - env.pop("USE_DDTRACE", None) - env.pop("DD_TRACE_OPENAI_ENABLED", None) - if use_ddtrace is not None: - env["USE_DDTRACE"] = use_ddtrace - result = subprocess.run( - ["sh", str(script), *argv], - env=env, - capture_output=True, - text=True, - check=False, - ) + +def _run_entrypoint( + argv: tuple[str, ...], + tmp_path: Path, + use_ddtrace: str | None = None, + **overrides: str | None, +) -> tuple[str, ...]: + """Run the entrypoint with stubbed executables on PATH and return the recorded lines.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir(parents=True) + _write_stubs(bin_dir, STUBBED_EXECUTABLES) + record = tmp_path / "record.txt" + env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace, **overrides}) + + result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT), *argv], env=env, capture_output=True, text=True, check=False) assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" return tuple(record.read_text().splitlines()) if record.exists() else () def _run_shell_command(command: str, bin_dir: Path, record: Path, use_ddtrace: str | None) -> tuple[str, ...]: """Run a resolved Terraform launch command through `sh -c` and return the recorded lines.""" - env = { - **os.environ, - "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", - "RECORD": str(record), - "PYTHONPATH": PYTHONPATH_SENTINEL, - } - env.pop("USE_DDTRACE", None) - env.pop("DD_TRACE_OPENAI_ENABLED", None) - if use_ddtrace is not None: - env["USE_DDTRACE"] = use_ddtrace - + env = _entrypoint_env(bin_dir, record, {"USE_DDTRACE": use_ddtrace}) result = subprocess.run(["sh", "-c", command], env=env, capture_output=True, text=True, check=False) assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" return tuple(record.read_text().splitlines()) if record.exists() else () @@ -131,29 +114,137 @@ def _resolve_tf_local(terraform_file: Path, name: str) -> str: def _entrypoint_argv(dockerfile: Path) -> tuple[str, ...]: matches = _ENTRYPOINT_RE.findall(dockerfile.read_text()) assert matches, f"no exec-form ENTRYPOINT found in {dockerfile}" - parsed = json.loads(matches[-1]) - return tuple(str(part) for part in parsed) - - -def _cmd_argv(dockerfile: Path) -> tuple[str, ...]: - matches = _CMD_RE.findall(dockerfile.read_text()) - assert matches, f"no exec-form CMD found in {dockerfile}" return tuple(str(part) for part in json.loads(matches[-1])) -def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration( - tmp_path: Path, +@pytest.mark.parametrize( + "argv, expected_exec, expected_args", + [ + (("proxy", "--config", "/app/config.yaml"), "exec=litellm", "args=--config /app/config.yaml"), + (("gateway",), "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"), + ( + ("gateway", "--port", "8080"), + "exec=python", + "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 8080", + ), + (("backend",), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"), + (("backend", "--port", "9001"), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001 --port 9001"), + (("ui",), "exec=nginx", "args=-g daemon off;"), + (("migrations",), "exec=python", "args=/app/migrations/run.py"), + (("metrics", "--port", "9090"), "exec=python", "args=-m litellm.proxy.prometheus_metrics_server --port 9090"), + (("collector",), "exec=python", "args=-m litellm.proxy.collector"), + ], +) +def test_the_first_argument_selects_the_component( + argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path ) -> None: + recorded = _run_entrypoint(argv, tmp_path) + + assert recorded[:2] == (expected_exec, expected_args) + + +@pytest.mark.parametrize( + "argv, expected", + [ + ((), ("exec=litellm", "args=")), + (("--port", "4000"), ("exec=litellm", "args=--port 4000")), + ( + ("--config", "/app/config.yaml", "--detailed_debug"), + ("exec=litellm", "args=--config /app/config.yaml --detailed_debug"), + ), + ], +) +def test_flags_alone_still_run_the_monolithic_proxy( + argv: tuple[str, ...], expected: tuple[str, str], tmp_path: Path +) -> None: + """Every `docker run litellm --config ...` written before components existed keeps working.""" + assert _run_entrypoint(argv, tmp_path)[:2] == expected + + +def test_the_dockerfile_leaves_the_command_empty_so_the_env_var_can_pick_the_component(tmp_path: Path) -> None: + """A CMD naming a component would beat LITELLM_COMPONENT, and the proxy already listens on 4000 with no flags.""" + dockerfile_text = DOCKERFILE.read_text() + assert _entrypoint_argv(DOCKERFILE) == (IMAGE_ENTRYPOINT_PATH,), "the image ENTRYPOINT must be the bare dispatcher" + assert not _CMD_RE.search(dockerfile_text), "the Dockerfile must not set a CMD" + + assert _run_entrypoint((), tmp_path)[:2] == ("exec=litellm", "args=") + + +@pytest.mark.parametrize( + "component, argv, expected_exec, expected_args", + [ + ("backend", (), "exec=uvicorn", "args=backend.main:app --host 0.0.0.0 --port 4001"), + ( + "gateway", + ("--port", "4100"), + "exec=python", + "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000 --port 4100", + ), + ("ui", (), "exec=nginx", "args=-g daemon off;"), + ], +) +def test_litellm_component_env_selects_the_component_when_the_command_has_none( + component: str, argv: tuple[str, ...], expected_exec: str, expected_args: str, tmp_path: Path +) -> None: + """Helm and Compose can pick the component with an env var and keep `args` for the process flags.""" + recorded = _run_entrypoint(argv, tmp_path, LITELLM_COMPONENT=component) + + assert recorded[:2] == (expected_exec, expected_args) + + +def test_an_explicit_component_argument_beats_the_env_var(tmp_path: Path) -> None: + recorded = _run_entrypoint(("backend",), tmp_path, LITELLM_COMPONENT="gateway") + + assert recorded[0] == "exec=uvicorn" + + +def test_the_gateway_honours_num_workers(tmp_path: Path) -> None: + recorded = _run_entrypoint(("gateway",), tmp_path, NUM_WORKERS="4") + + assert recorded[1] == "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000" + + +def test_an_unknown_first_word_is_run_as_the_command(tmp_path: Path) -> None: + """`docker run sh -c ...` and `kubectl exec`-style overrides bypass the dispatcher.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + _write_stubs(bin_dir, ("some-tool",)) + record = tmp_path / "record.txt" + env = _entrypoint_env(bin_dir, record, {}) + + result = subprocess.run( + ["sh", str(DOCKER_ENTRYPOINT), "some-tool", "--flag"], env=env, capture_output=True, text=True, check=False + ) + + assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" + assert record.read_text().splitlines()[:2] == ["exec=some-tool", "args=--flag"] + + +def test_an_unknown_litellm_component_fails_fast_instead_of_guessing(tmp_path: Path) -> None: + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + _write_stubs(bin_dir, STUBBED_EXECUTABLES) + record = tmp_path / "record.txt" + env = _entrypoint_env(bin_dir, record, {"LITELLM_COMPONENT": "gatway"}) + + result = subprocess.run(["sh", str(DOCKER_ENTRYPOINT)], env=env, capture_output=True, text=True, check=False) + + assert result.returncode == 64 + assert "gatway" in result.stderr + assert not record.exists(), "nothing may be exec'd when the component is unknown" + + +def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration(tmp_path: Path) -> None: """`USE_DDTRACE=true` must prefix the command with `ddtrace-run`, turn the openai integration off, and leave PYTHONPATH alone. `ddtrace-run` installs its instrumentation by PREPENDING a bootstrap directory to - PYTHONPATH, and the images set PYTHONPATH=/app so the app package is importable. A - wrapper that assigned PYTHONPATH instead of inheriting it would either drop the - bootstrap (silently disabling tracing) or drop /app (breaking the import), so the - recorded value is asserted verbatim. + PYTHONPATH, and the image sets PYTHONPATH=/app so the component packages are + importable. A wrapper that assigned PYTHONPATH instead of inheriting it would either + drop the bootstrap (silently disabling tracing) or drop /app (breaking the import), so + the recorded value is asserted verbatim. - PYTHONPATH_SENTINEL deliberately differs from the images' own /app: with /app as the + PYTHONPATH_SENTINEL deliberately differs from the image's own /app: with /app as the fixture value, a wrapper that overwrote PYTHONPATH with /app would still satisfy this assertion and the check would prove nothing. @@ -161,32 +252,22 @@ def test_ddtrace_enabled_wraps_the_command_and_disables_the_openai_integration( it before any litellm code runs, so litellm's in-process `patch_all(..., openai=False)` can no longer suppress it; leaving it on double-reports every LLM call. """ - recorded = _run_entrypoint( - COMPONENT_ENTRYPOINT, - ("uvicorn", "gateway.main:app", "--workers", "2", "--port", "4000"), - use_ddtrace="true", - tmp_path=tmp_path, - ) + recorded = _run_entrypoint(("gateway",), tmp_path, use_ddtrace="true", NUM_WORKERS="2") assert recorded == ( "exec=ddtrace-run", - "args=uvicorn gateway.main:app --workers 2 --port 4000", + "args=python -m gateway.launch --workers 2 --host 0.0.0.0 --port 4000", "DD_TRACE_OPENAI_ENABLED=False", f"PYTHONPATH={PYTHONPATH_SENTINEL}", ) def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None: - recorded = _run_entrypoint( - COMPONENT_ENTRYPOINT, - ("uvicorn", "backend.main:app", "--port", "4001"), - use_ddtrace=None, - tmp_path=tmp_path, - ) + recorded = _run_entrypoint(("backend", "--workers", "2"), tmp_path) assert recorded == ( "exec=uvicorn", - "args=backend.main:app --port 4001", + "args=backend.main:app --host 0.0.0.0 --port 4001 --workers 2", "DD_TRACE_OPENAI_ENABLED=", f"PYTHONPATH={PYTHONPATH_SENTINEL}", ) @@ -196,35 +277,24 @@ def test_ddtrace_disabled_execs_the_command_directly(tmp_path: Path) -> None: "use_ddtrace, traced", [*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)], ) -def test_gating_matches_the_monolithic_entrypoint_and_get_secret_bool( +def test_ddtrace_gating_matches_get_secret_bool_for_every_component( use_ddtrace: str | None, traced: bool, tmp_path: Path ) -> None: - """Both entrypoints must accept exactly the spellings `get_secret_bool` accepts. + """The shell gate must accept exactly the spellings `get_secret_bool` accepts. `ProxyStartupEvent._init_dd_tracer` reads `USE_DDTRACE` through `get_secret_bool`, which matches `true` case-insensitively. If the shell gate were stricter, `USE_DDTRACE=True` would give in-process LLM spans without `ddtrace-run` HTTP spans, a half-enabled state. """ - component = _run_entrypoint( - COMPONENT_ENTRYPOINT, - ("uvicorn", "gateway.main:app"), - use_ddtrace=use_ddtrace, - tmp_path=tmp_path / "component", - ) - monolith = _run_entrypoint( - PROD_ENTRYPOINT, - ("--port", "4000"), - use_ddtrace=use_ddtrace, - tmp_path=tmp_path / "monolith", - ) + component = _run_entrypoint(("gateway",), tmp_path / "component", use_ddtrace=use_ddtrace) + monolith = _run_entrypoint(("--port", "4000"), tmp_path / "monolith", use_ddtrace=use_ddtrace) - expected_exec = "exec=ddtrace-run" if traced else "exec=uvicorn" expected_openai = "DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=" - assert component[0] == expected_exec + assert component[0] == ("exec=ddtrace-run" if traced else "exec=python") assert component[2] == expected_openai assert monolith[0] == ("exec=ddtrace-run" if traced else "exec=litellm") - assert monolith[2] == expected_openai assert monolith[1] == ("args=litellm --port 4000" if traced else "args=--port 4000") + assert monolith[2] == expected_openai def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path) -> None: @@ -236,19 +306,26 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path) (multiproc_dir / "counter_7.db").write_bytes(b"stale") (multiproc_dir / "keep.txt").write_text("not a sample") + recorded = _run_entrypoint(("gateway",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir)) + + assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"] + assert recorded[0] == "exec=python" + + +def test_a_raw_command_leaves_the_prometheus_multiproc_dir_untouched(tmp_path: Path) -> None: + """The entrypoint cannot tell a raw writer from a raw reader, so `docker run python -m + litellm.proxy.prometheus_metrics_server` must not delete the samples the gateway workers are still writing.""" + multiproc_dir = tmp_path / "multiproc" + multiproc_dir.mkdir() + (multiproc_dir / "counter_7.db").write_bytes(b"live") bin_dir = tmp_path / "bin" bin_dir.mkdir() - _write_stubs(bin_dir, ("uvicorn",)) + _write_stubs(bin_dir, ("python",)) record = tmp_path / "record.txt" - env = { - **os.environ, - "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", - "RECORD": str(record), - "PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir), - } - env.pop("USE_DDTRACE", None) + env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)}) + result = subprocess.run( - ["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"], + ["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "litellm.proxy.prometheus_metrics_server"], env=env, capture_output=True, text=True, @@ -256,161 +333,69 @@ def test_wipes_the_prometheus_multiproc_dir_before_uvicorn_forks(tmp_path: Path) ) assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" - assert sorted(p.name for p in multiproc_dir.iterdir()) == ["keep.txt"] - assert record.read_text().splitlines()[0] == "exec=uvicorn" + assert [p.name for p in multiproc_dir.iterdir()] == ["counter_7.db"] + assert record.read_text().splitlines()[0] == "exec=python" + + +def test_a_raw_command_still_gets_the_prometheus_multiproc_dir_created(tmp_path: Path) -> None: + """`docker run python -m gateway.launch` cannot write samples into a directory that is not there.""" + multiproc_dir = tmp_path / "not-yet" / "multiproc" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + _write_stubs(bin_dir, ("python",)) + record = tmp_path / "record.txt" + env = _entrypoint_env(bin_dir, record, {"PROMETHEUS_MULTIPROC_DIR": str(multiproc_dir)}) + + result = subprocess.run( + ["sh", str(DOCKER_ENTRYPOINT), "python", "-m", "gateway.launch"], + env=env, + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" + assert multiproc_dir.is_dir() + assert record.read_text().splitlines()[0] == "exec=python" + + +@pytest.mark.parametrize("component", ["metrics", "collector"]) +def test_readers_of_the_prometheus_multiproc_dir_keep_the_workers_samples(component: str, tmp_path: Path) -> None: + """The metrics and collector sidecars share the volume with gateway workers that are already + serving traffic, so their (re)start must not erase the samples those workers have written.""" + multiproc_dir = tmp_path / "multiproc" + multiproc_dir.mkdir() + (multiproc_dir / "counter_12.db").write_bytes(b"live") + (multiproc_dir / "histogram_13.db").write_bytes(b"live") + + recorded = _run_entrypoint((component,), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(multiproc_dir)) + + assert sorted(p.name for p in multiproc_dir.iterdir()) == ["counter_12.db", "histogram_13.db"] + assert recorded[0] == "exec=python" def test_creates_a_missing_prometheus_multiproc_dir(tmp_path: Path) -> None: - bin_dir = tmp_path / "bin" - bin_dir.mkdir() - _write_stubs(bin_dir, ("uvicorn",)) missing = tmp_path / "multiproc" - env = { - **os.environ, - "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", - "RECORD": str(tmp_path / "record.txt"), - "PROMETHEUS_MULTIPROC_DIR": str(missing), - } - env.pop("USE_DDTRACE", None) - result = subprocess.run( - ["sh", str(COMPONENT_ENTRYPOINT), "uvicorn", "gateway.main:app"], env=env, capture_output=True, text=True - ) - assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" + _run_entrypoint(("backend",), tmp_path, PROMETHEUS_MULTIPROC_DIR=str(missing)) + assert missing.is_dir() -def _copied_script(dockerfile: Path, image_path: str) -> Path: - """Resolve the repo file a Dockerfile `COPY`s to `image_path`, so tests run what the image ships.""" - matches = _COPY_RE.findall(dockerfile.read_text()) - sources = tuple(src for src, dst in matches if dst == image_path) - assert sources, f"{dockerfile} never COPYs anything to {image_path}" - source = REPO_ROOT / sources[-1] - assert source.is_file(), f"{dockerfile} COPYs {sources[-1]}, which does not exist in the build context" - return source - - -@pytest.mark.parametrize( - "use_ddtrace, traced", - [*((v, True) for v in TRUTHY_USE_DDTRACE), *((v, False) for v in FALSY_USE_DDTRACE)], -) -def test_build_from_pip_image_launches_litellm_through_the_prod_entrypoint( - use_ddtrace: str | None, traced: bool, tmp_path: Path -) -> None: - """Run the build_from_pip image's ENTRYPOINT + CMD through the script it actually COPYs. - - The image used to `ENTRYPOINT ["litellm"]`, so `USE_DDTRACE` was inert there at any spelling. - Resolving the ENTRYPOINT path back to its COPY source and executing it with the Dockerfile's - CMD checks the launch the container performs, not just that the Dockerfile mentions the script. - """ - entrypoint = _entrypoint_argv(BUILD_FROM_PIP_DOCKERFILE) - assert len(entrypoint) == 1, ( - f"{BUILD_FROM_PIP_DOCKERFILE} ENTRYPOINT must be the bare script so CMD reaches litellm" - ) - script = _copied_script(BUILD_FROM_PIP_DOCKERFILE, entrypoint[0]) - assert script == PROD_ENTRYPOINT, f"{BUILD_FROM_PIP_DOCKERFILE} bypasses the ddtrace-aware entrypoint" - assert f"chmod +x {entrypoint[0]}" in BUILD_FROM_PIP_DOCKERFILE.read_text() - - cmd = _cmd_argv(BUILD_FROM_PIP_DOCKERFILE) - recorded = _run_entrypoint(script, cmd, use_ddtrace=use_ddtrace, tmp_path=tmp_path) - - cmd_str = " ".join(cmd) - assert recorded == ( - "exec=ddtrace-run" if traced else "exec=litellm", - f"args=litellm {cmd_str}" if traced else f"args={cmd_str}", - "DD_TRACE_OPENAI_ENABLED=False" if traced else "DD_TRACE_OPENAI_ENABLED=", - f"PYTHONPATH={PYTHONPATH_SENTINEL}", - ) - - -def test_entrypoint_script_is_executable() -> None: - mode = COMPONENT_ENTRYPOINT.stat().st_mode - assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT" - assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user" - - -def test_entrypoint_script_has_no_carriage_returns() -> None: - assert b"\r" not in COMPONENT_ENTRYPOINT.read_bytes() - - -@pytest.mark.parametrize( - "dockerfile, launcher", - [ - (GATEWAY_DOCKERFILE, "python -m gateway.launch"), - (BACKEND_DOCKERFILE, "uvicorn backend.main:app"), - ], -) -def test_component_images_launch_uvicorn_through_the_entrypoint(dockerfile: Path, launcher: str) -> None: - entrypoint = " ".join(_entrypoint_argv(dockerfile)) - - assert IMAGE_ENTRYPOINT_PATH in entrypoint, f"{dockerfile} bypasses the ddtrace-aware entrypoint" - assert launcher in entrypoint - assert entrypoint.index(IMAGE_ENTRYPOINT_PATH) < entrypoint.index(launcher), ( - f"{dockerfile} must invoke uvicorn through the entrypoint, not the other way around" - ) - - -@pytest.mark.parametrize( - "use_ddtrace, num_workers, expected_exec, expected_args", - [ - (None, "4", "exec=python", "args=-m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"), - (None, None, "exec=python", "args=-m gateway.launch --workers 1 --host 0.0.0.0 --port 4000"), - ("true", "4", "exec=ddtrace-run", "args=python -m gateway.launch --workers 4 --host 0.0.0.0 --port 4000"), - ], -) -def test_gateway_image_execs_the_supervisor_with_its_worker_count( - use_ddtrace: str | None, num_workers: str | None, expected_exec: str, expected_args: str, tmp_path: Path -) -> None: - """Run the gateway image's ENTRYPOINT + CMD and record what the container execs. - - The Dockerfile's `/app/...` script path is resolved to the checked-in script and `python` - is stubbed on PATH, so the assertion is on the argv `gateway.launch` receives, not on the - Dockerfile text. - """ - entrypoint = tuple( - part.replace(IMAGE_ENTRYPOINT_PATH, str(COMPONENT_ENTRYPOINT)) for part in _entrypoint_argv(GATEWAY_DOCKERFILE) - ) - bin_dir = tmp_path / "bin" - bin_dir.mkdir(parents=True) - _write_stubs(bin_dir, ("ddtrace-run", "python", "uvicorn")) - record = tmp_path / "record.txt" - overrides = {"USE_DDTRACE": use_ddtrace, "NUM_WORKERS": num_workers} - env = { - **{k: v for k, v in os.environ.items() if k not in ("DD_TRACE_OPENAI_ENABLED", *overrides)}, - **{k: v for k, v in overrides.items() if v is not None}, - "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", - "RECORD": str(record), - "PYTHONPATH": PYTHONPATH_SENTINEL, - } - - result = subprocess.run( - [*entrypoint, *_cmd_argv(GATEWAY_DOCKERFILE)], env=env, capture_output=True, text=True, check=False - ) - - assert result.returncode == 0, f"stdout={result.stdout} stderr={result.stderr}" - assert tuple(record.read_text().splitlines())[:2] == (expected_exec, expected_args) - - -@pytest.mark.parametrize("dockerfile", [GATEWAY_DOCKERFILE, BACKEND_DOCKERFILE]) -def test_component_images_make_the_entrypoint_executable(dockerfile: Path) -> None: - body = dockerfile.read_text() - assert "chmod +x docker/component_entrypoint.sh" in body - - @pytest.mark.parametrize("terraform_file", TERRAFORM_LAUNCH_SITES, ids=lambda p: p.parent.name) @pytest.mark.parametrize("component", ["gateway", "backend"]) @pytest.mark.parametrize("use_ddtrace", [*TRUTHY_USE_DDTRACE, *FALSY_USE_DDTRACE]) def test_terraform_launch_command_matches_the_script_contract( terraform_file: Path, component: str, use_ddtrace: str | None, tmp_path: Path ) -> None: - """The Terraform command and `docker/component_entrypoint.sh` must decide identically. + """The Terraform command and `docker-entrypoint.sh ` must decide identically. The decision deliberately lives in two places. The script is what the image ENTRYPOINT runs; the Terraform strings are what runs when a deployment overrides that ENTRYPOINT, and they cannot call the script because the caller supplies the image tag and it may predate the file. - Both modules default to a tag that does. So instead of asserting a shared path, this runs both - implementations under the same environment and asserts they agree on which binary is exec'd - and on whether the openai integration is disabled. + So instead of asserting a shared path, this runs both implementations under the same + environment and asserts they agree on which binary is exec'd and on whether the openai + integration is disabled. """ launcher = COMPONENT_LAUNCHERS[component] app_target = " ".join(launcher[1:]) @@ -421,18 +406,14 @@ def test_terraform_launch_command_matches_the_script_contract( _write_stubs(bin_dir, ("ddtrace-run", "uvicorn", "python")) from_terraform = _run_shell_command(command, bin_dir, tmp_path / "terraform.txt", use_ddtrace) - from_script = _run_entrypoint( - COMPONENT_ENTRYPOINT, - launcher, - use_ddtrace=use_ddtrace, - tmp_path=tmp_path / "script", - ) + from_script = _run_entrypoint((component,), tmp_path / "script", use_ddtrace=use_ddtrace) assert from_terraform[0] == from_script[0], ( f"{terraform_file} disagrees with the script on USE_DDTRACE={use_ddtrace}" ) assert from_terraform[2] == from_script[2], f"{terraform_file} disagrees with the script on the openai integration" assert app_target in from_terraform[1] + assert app_target in from_script[1] assert "gateway.main:app" not in from_terraform[1], f"{terraform_file} bypasses the gateway.launch supervisor" if use_ddtrace in TRUTHY_USE_DDTRACE: @@ -478,9 +459,11 @@ def test_terraform_does_not_depend_on_the_entrypoint_script(terraform_file: Path ) -def test_gateway_keeps_its_worker_count_and_backend_keeps_a_single_process() -> None: - gateway = " ".join(_entrypoint_argv(GATEWAY_DOCKERFILE)) + " " + " ".join(_cmd_argv(GATEWAY_DOCKERFILE)) - backend = " ".join(_entrypoint_argv(BACKEND_DOCKERFILE)) + " " + " ".join(_cmd_argv(BACKEND_DOCKERFILE)) +def test_entrypoint_script_is_executable() -> None: + mode = DOCKER_ENTRYPOINT.stat().st_mode + assert mode & stat.S_IXUSR, "entrypoint must be committed executable to run as the image ENTRYPOINT" + assert mode & stat.S_IXOTH, "entrypoint must be executable by the unprivileged `nonroot` user" - assert "--workers" in gateway and "NUM_WORKERS" in gateway - assert "--workers" not in backend and "NUM_WORKERS" not in backend + +def test_entrypoint_script_has_no_carriage_returns() -> None: + assert b"\r" not in DOCKER_ENTRYPOINT.read_bytes() diff --git a/tests/unit/test_dockerfile_bedrock_realtime_extra.py b/tests/unit/test_dockerfile_bedrock_realtime_extra.py index e157c982105..f4863b5ff77 100644 --- a/tests/unit/test_dockerfile_bedrock_realtime_extra.py +++ b/tests/unit/test_dockerfile_bedrock_realtime_extra.py @@ -1,5 +1,5 @@ """ -Static checks that every proxy Docker image installs the `bedrock-realtime` extra. +Static checks that the shipped Docker image installs the `bedrock-realtime` extra. Bedrock Nova Sonic speech-to-speech (`/v1/realtime`) needs `aws-sdk-bedrock-runtime`, which only ships in the `bedrock-realtime` extra. An image whose `uv sync` stages @@ -23,12 +23,7 @@ else: REPO_ROOT: Final = os.path.join(os.path.dirname(__file__), "..", "..") -PROXY_DOCKERFILES: Final = ( - "Dockerfile", - os.path.join("docker", "Dockerfile.non_root"), - os.path.join("docker", "Dockerfile.database"), - os.path.join("gateway", "Dockerfile"), -) +PROXY_DOCKERFILES: Final = ("Dockerfile",) CONTINUED_LINE_RE: Final = re.compile(r"(?:\\\n|[^\n])+") UV_SYNC_BOUNDARY_RE: Final = re.compile(r"(?=uv sync)") diff --git a/tests/unit/test_dockerfile_non_root.py b/tests/unit/test_dockerfile_non_root.py index 694da6368e7..d37d06c337d 100644 --- a/tests/unit/test_dockerfile_non_root.py +++ b/tests/unit/test_dockerfile_non_root.py @@ -1,39 +1,26 @@ """ -Static checks on docker/Dockerfile.non_root. +Static checks on the shipped Dockerfile's runtime user. -The non_root image is intended for deployment into hardened Kubernetes -clusters where `securityContext.runAsNonRoot: true` is enforced. The -kubelet validates non-root status by parsing the image's USER field as -an integer — a string name like "nobody" is rejected with -CreateContainerConfigError because the kubelet cannot resolve -/etc/passwd inside the image at admission time. +The image is deployed into hardened Kubernetes clusters where +`securityContext.runAsNonRoot: true` is enforced. The kubelet validates +non-root status by parsing the image's USER field as an integer: a string +name like "nonroot" is rejected with CreateContainerConfigError because the +kubelet cannot resolve /etc/passwd inside the image at admission time. """ import os import re -import pytest - -DOCKERFILE_PATH = os.path.join( - os.path.dirname(__file__), - "..", - "..", - "docker", - "Dockerfile.non_root", -) +DOCKERFILE_PATH = os.path.join(os.path.dirname(__file__), "..", "..", "Dockerfile") def _final_user_directive(dockerfile_text: str) -> str: - """Return the value of the last `USER` directive in the file.""" + """Return the uid of the last `USER` directive in the file (`USER uid` or `USER uid:gid`).""" matches = re.findall(r"^USER\s+(\S+)\s*$", dockerfile_text, re.MULTILINE) - assert matches, "Dockerfile.non_root has no USER directive" - return matches[-1] + assert matches, "Dockerfile has no USER directive" + return matches[-1].split(":", 1)[0] -@pytest.mark.skipif( - not os.path.exists(DOCKERFILE_PATH), - reason="Dockerfile.non_root not present in this checkout", -) def test_final_user_directive_is_numeric(): """The runtime USER must be a numeric UID so kubelet's runAsNonRoot admission check (strconv.Atoi) succeeds.""" @@ -43,12 +30,9 @@ def test_final_user_directive_is_numeric(): final_user = _final_user_directive(contents) assert final_user.isdigit(), ( - f"Dockerfile.non_root final USER is {final_user!r}; must be a numeric UID " + f"Dockerfile final USER is {final_user!r}; must be a numeric UID " "so Kubernetes' runAsNonRoot admission check can verify non-root status. " "See https://kubernetes.io/docs/tasks/configure-pod-container/security-context/" ) - assert int(final_user) != 0, ( - f"Dockerfile.non_root final USER is {final_user} (root); the non_root image " - "must run as a non-zero UID." - ) + assert int(final_user) != 0, f"Dockerfile final USER is {final_user} (root); the image must run as a non-zero UID." diff --git a/ui/Dockerfile b/ui/Dockerfile deleted file mode 100644 index f14b631685a..00000000000 --- a/ui/Dockerfile +++ /dev/null @@ -1,42 +0,0 @@ -# syntax=docker/dockerfile:1.7 - -# UI container — Next.js static export served by nginx. - -ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43 -ARG NGINX_VERSION=1.31.5-alpine3.24@sha256:34f40471dea485273c5e2a04dd5e97a682332ceb4a9adecd67de450dcb2fb390 - -# ---------- builder ---------- -FROM ${UI_BUILD_IMAGE} AS builder - -ENV NEXT_TELEMETRY_DISABLED=1 \ - npm_config_fund=false \ - npm_config_audit=false - -WORKDIR /app - -# Layer the lockfile-only install above the source copy so source-only -# edits don't bust the install cache. -COPY ui/litellm-dashboard/package.json ui/litellm-dashboard/package-lock.json ./ -RUN --mount=type=cache,target=/root/.npm \ - npm ci --prefer-offline - -COPY ui/litellm-dashboard/ ./ -RUN npm run build - -# ---------- runtime ---------- -FROM nginx:${NGINX_VERSION} AS runtime - -# Drop the upstream default :80 server; we own the config. -RUN rm -f /etc/nginx/conf.d/default.conf - -# Static export → web root. -COPY --from=builder /app/out /usr/share/nginx/html - -# Routing rules — see ui/nginx.conf for the full description. -COPY ui/nginx.conf /etc/nginx/nginx.conf - -EXPOSE 3000/tcp - -# nginx as PID 1 in foreground; respects SIGTERM out of the box, so -# no tini/dumb-init wrapper needed. -CMD ["nginx", "-g", "daemon off;"] diff --git a/ui/nginx.conf b/ui/nginx.conf index 235cb9c501e..db4db9116a5 100644 --- a/ui/nginx.conf +++ b/ui/nginx.conf @@ -5,6 +5,7 @@ worker_processes auto; # nginx image's /var/cache/nginx and /run are root-owned 755) and works # with readOnlyRootFilesystem when /tmp is an emptyDir. pid /tmp/nginx.pid; +error_log /dev/stderr warn; events { worker_connections 1024; } @@ -15,6 +16,7 @@ http { uwsgi_temp_path /tmp/nginx-uwsgi-temp; scgi_temp_path /tmp/nginx-scgi-temp; + access_log /dev/stdout; include /etc/nginx/mime.types; default_type application/octet-stream; sendfile on; @@ -29,7 +31,6 @@ http { application/javascript application/json text/css - text/html image/svg+xml font/woff font/woff2; @@ -37,16 +38,16 @@ http { server { listen 3000 default_server; server_name _; - root /usr/share/nginx/html; + root /var/lib/litellm/ui; # next.config.mjs sets assetPrefix=/litellm-asset-prefix, which makes # the built HTML reference /litellm-asset-prefix/_next/... — but the # static export only emits files under /_next/. Map the prefix to # the real tree at request time instead of duplicating the directory # at build time. NB: alias rewrites the location prefix, so - # /litellm-asset-prefix/_next/foo.js → /usr/share/nginx/html/_next/foo.js. + # /litellm-asset-prefix/_next/foo.js → /var/lib/litellm/ui/_next/foo.js. location /litellm-asset-prefix/_next/ { - alias /usr/share/nginx/html/_next/; + alias /var/lib/litellm/ui/_next/; expires 1y; add_header Cache-Control "public, immutable"; } @@ -63,7 +64,7 @@ http { add_header Cache-Control "public, immutable"; } location ^~ /ui/assets/ { - alias /usr/share/nginx/html/assets/; + alias /var/lib/litellm/ui/assets/; } location = /favicon.ico { try_files $uri =404;