diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml
index 8fbf1b3c5b4..39b46cba999 100644
--- a/.github/ISSUE_TEMPLATE/bug_report.yml
+++ b/.github/ISSUE_TEMPLATE/bug_report.yml
@@ -23,13 +23,15 @@ body:
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
render: shell
- type: dropdown
- id: ml-ops-team
+ id: component
attributes:
- label: Are you a ML Ops Team?
- description: This helps us prioritize your requests correctly
+ label: What part of LiteLLM is this about?
options:
- - "No"
- - "Yes"
+ - "SDK (litellm Python package)"
+ - "Proxy"
+ - "UI Dashboard"
+ - "Docs"
+ - "Other"
validations:
required: true
- type: input
diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml
index 13a2132ec95..96b95cc7f02 100644
--- a/.github/ISSUE_TEMPLATE/feature_request.yml
+++ b/.github/ISSUE_TEMPLATE/feature_request.yml
@@ -22,6 +22,18 @@ body:
description: Please outline the motivation for the proposal. Is your feature request related to a specific problem? e.g., "I'm working on X and would like Y to be possible". If this is related to another GitHub issue, please link here too.
validations:
required: true
+ - type: dropdown
+ id: component
+ attributes:
+ label: What part of LiteLLM is this about?
+ options:
+ - "SDK (litellm Python package)"
+ - "Proxy"
+ - "UI Dashboard"
+ - "Docs"
+ - "Other"
+ validations:
+ required: true
- type: dropdown
id: hiring-interest
attributes:
diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md
index 8977332ee01..b91b16c955c 100644
--- a/.github/pull_request_template.md
+++ b/.github/pull_request_template.md
@@ -1,7 +1,3 @@
-## Title
-
-
-
## Relevant issues
@@ -11,7 +7,6 @@
**Please complete all items before asking a LiteLLM maintainer to review your PR**
- [ ] I have Added testing in the [`tests/litellm/`](https://github.com/BerriAI/litellm/tree/main/tests/litellm) directory, **Adding at least 1 test is a hard requirement** - [see details](https://docs.litellm.ai/docs/extras/contributing_code)
-- [ ] I have added a screenshot of my new test passing locally
- [ ] My PR passes all unit tests on [`make test-unit`](https://docs.litellm.ai/docs/extras/contributing_code)
- [ ] My PR's scope is as isolated as possible, it only solves 1 specific problem
diff --git a/.github/workflows/create_daily_staging_branch.yml b/.github/workflows/create_daily_staging_branch.yml
new file mode 100644
index 00000000000..a97cf6f9740
--- /dev/null
+++ b/.github/workflows/create_daily_staging_branch.yml
@@ -0,0 +1,43 @@
+name: Create Daily Staging Branch
+
+on:
+ schedule:
+ - cron: '0 0 * * *' # Runs daily at midnight UTC
+ workflow_dispatch: # Allow manual trigger
+
+jobs:
+ create-staging-branch:
+ runs-on: ubuntu-latest
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v3
+ with:
+ fetch-depth: 0
+
+ - name: Create daily staging branch
+ env:
+ GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: |
+ # Configure Git user
+ git config user.name "github-actions[bot]"
+ git config user.email "github-actions[bot]@users.noreply.github.com"
+
+ # Generate branch name with MM_DD_YYYY format
+ BRANCH_NAME="litellm_staging_$(date +'%m_%d_%Y')"
+ echo "Creating branch: $BRANCH_NAME"
+
+ # Fetch all branches
+ git fetch --all
+
+ # Check if the branch already exists
+ if git show-ref --verify --quiet refs/remotes/origin/$BRANCH_NAME; then
+ echo "Branch $BRANCH_NAME already exists. Skipping creation."
+ else
+ echo "Creating new branch: $BRANCH_NAME"
+ # Create the new branch from main
+ git checkout -b $BRANCH_NAME origin/main
+ # Push the new branch
+ git push origin $BRANCH_NAME
+ echo "Successfully created and pushed branch: $BRANCH_NAME"
+ fi
diff --git a/.github/workflows/issue-keyword-labeler.yml b/.github/workflows/issue-keyword-labeler.yml
index 60c18e3b9af..936f90f747f 100644
--- a/.github/workflows/issue-keyword-labeler.yml
+++ b/.github/workflows/issue-keyword-labeler.yml
@@ -19,7 +19,7 @@ jobs:
id: scan
env:
PROVIDER_ISSUE_WEBHOOK_URL: ${{ secrets.PROVIDER_ISSUE_WEBHOOK_URL }}
- KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic
+ KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic,gemini,cohere,mistral,groq,ollama,deepseek
run: python3 .github/scripts/scan_keywords.py
- name: Ensure label exists
diff --git a/.github/workflows/label-component.yml b/.github/workflows/label-component.yml
new file mode 100644
index 00000000000..c0f9436288c
--- /dev/null
+++ b/.github/workflows/label-component.yml
@@ -0,0 +1,144 @@
+name: Label Component Issues
+
+on:
+ issues:
+ types:
+ - opened
+
+jobs:
+ add-component-label:
+ runs-on: ubuntu-latest
+ permissions:
+ issues: write
+ steps:
+ - name: Add SDK label
+ if: contains(github.event.issue.body, 'SDK (litellm Python package)')
+ uses: actions/github-script@v7
+ with:
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ script: |
+ const labelName = 'sdk';
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName,
+ color: '0E7C86',
+ description: 'Issues related to the litellm Python SDK'
+ });
+ } else {
+ throw error;
+ }
+ }
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [labelName]
+ });
+
+ - name: Add Proxy label
+ if: contains(github.event.issue.body, 'Proxy')
+ uses: actions/github-script@v7
+ with:
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ script: |
+ const labelName = 'proxy';
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName,
+ color: '5319E7',
+ description: 'Issues related to the LiteLLM Proxy'
+ });
+ } else {
+ throw error;
+ }
+ }
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [labelName]
+ });
+
+ - name: Add UI Dashboard label
+ if: contains(github.event.issue.body, 'UI Dashboard')
+ uses: actions/github-script@v7
+ with:
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ script: |
+ const labelName = 'ui-dashboard';
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName,
+ color: 'D876E3',
+ description: 'Issues related to the LiteLLM UI Dashboard'
+ });
+ } else {
+ throw error;
+ }
+ }
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [labelName]
+ });
+
+ - name: Add Docs label
+ if: contains(github.event.issue.body, 'Docs')
+ uses: actions/github-script@v7
+ with:
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ script: |
+ const labelName = 'docs';
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: labelName,
+ color: 'FBCA04',
+ description: 'Issues related to LiteLLM documentation'
+ });
+ } else {
+ throw error;
+ }
+ }
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [labelName]
+ });
diff --git a/.github/workflows/label-mlops.yml b/.github/workflows/label-mlops.yml
deleted file mode 100644
index 37789c1ea76..00000000000
--- a/.github/workflows/label-mlops.yml
+++ /dev/null
@@ -1,17 +0,0 @@
-name: Label ML Ops Team Issues
-
-on:
- issues:
- types:
- - opened
-
-jobs:
- add-mlops-label:
- runs-on: ubuntu-latest
- steps:
- - name: Check if ML Ops Team is selected
- uses: actions-ecosystem/action-add-labels@v1
- if: contains(github.event.issue.body, '### Are you a ML Ops Team?') && contains(github.event.issue.body, 'Yes')
- with:
- github_token: ${{ secrets.GITHUB_TOKEN }}
- labels: "mlops user request"
diff --git a/deploy/charts/litellm-helm/README.md b/deploy/charts/litellm-helm/README.md
index 6fdc423a177..2fa856843f3 100644
--- a/deploy/charts/litellm-helm/README.md
+++ b/deploy/charts/litellm-helm/README.md
@@ -29,7 +29,7 @@ If `db.useStackgresOperator` is used (not yet implemented):
| `masterkey` | The Master API Key for LiteLLM. If not specified, a random key in the `sk-...` format is generated. | N/A |
| `environmentSecrets` | An optional array of Secret object names. The keys and values in these secrets will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
| `environmentConfigMaps` | An optional array of ConfigMap object names. The keys and values in these configmaps will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
-| `image.repository` | LiteLLM Proxy image repository | `ghcr.io/berriai/litellm` |
+| `image.repository` | LiteLLM Proxy image repository | `docker.litellm.ai/berriai/litellm` |
| `image.pullPolicy` | LiteLLM Proxy image pull policy | `IfNotPresent` |
| `image.tag` | Overrides the image tag whose default the latest version of LiteLLM at the time this chart was published. | `""` |
| `imagePullSecrets` | Registry credentials for the LiteLLM and initContainer images. | `[]` |
diff --git a/docker-compose.hardened.yml b/docker-compose.hardened.yml
new file mode 100644
index 00000000000..31d0c2e9ef2
--- /dev/null
+++ b/docker-compose.hardened.yml
@@ -0,0 +1,46 @@
+services:
+ # Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints.
+ # Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage.
+ litellm:
+ build:
+ context: .
+ dockerfile: docker/Dockerfile.non_root
+ target: runtime
+ args:
+ PROXY_EXTRAS_SOURCE: "local"
+ depends_on:
+ - squid
+ user: "101:101"
+ group_add:
+ - "2345"
+ read_only: true
+ cap_drop:
+ - ALL
+ security_opt:
+ - no-new-privileges:true
+ tmpfs:
+ - /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777
+ - /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777
+ volumes:
+ - ./proxy_server_config.yaml:/app/config.yaml:ro
+ environment:
+ LITELLM_NON_ROOT: "true"
+ PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries"
+ XDG_CACHE_HOME: "/app/cache"
+ LITELLM_MIGRATION_DIR: "/app/migrations"
+ HTTP_PROXY: "http://squid:3128"
+ HTTPS_PROXY: "http://squid:3128"
+ NO_PROXY: "localhost,127.0.0.1,db"
+ command:
+ - "--port"
+ - "4000"
+ - "--config"
+ - "/app/config.yaml"
+ squid:
+ image: sameersbn/squid:3.5.27-2
+ restart: unless-stopped
+ ports:
+ - "3128:3128"
+ tmpfs:
+ - /var/spool/squid:rw,noexec,nosuid,nodev,size=64m
+ - /var/log/squid:rw,noexec,nosuid,nodev,size=16m
diff --git a/docker-compose.yml b/docker-compose.yml
index 8898aff62da..988860a7877 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -4,7 +4,7 @@ services:
context: .
args:
target: runtime
- image: ghcr.io/berriai/litellm:main-stable
+ image: docker.litellm.ai/berriai/litellm:main-stable
#########################################
## Uncomment these lines to start proxy with a config.yaml file ##
# volumes:
diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root
index 9fc8acf2a18..d8a362680e4 100644
--- a/docker/Dockerfile.non_root
+++ b/docker/Dockerfile.non_root
@@ -1,154 +1,183 @@
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
+ARG PROXY_EXTRAS_SOURCE=published
# -----------------
# Builder Stage
# -----------------
FROM $LITELLM_BUILD_IMAGE AS builder
+ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
-
-# Install build dependencies including Node.js for UI build
USER root
+
+# Install build dependencies with retry logic (includes node for UI build)
RUN for i in 1 2 3; do \
- apk add --no-cache \
- python3 \
- py3-pip \
- clang \
- llvm \
- lld \
- gcc \
- linux-headers \
- build-base \
- bash \
- nodejs \
- npm && break || sleep 5; \
- done \
+ apk add --no-cache \
+ python3 \
+ py3-pip \
+ clang \
+ llvm \
+ lld \
+ gcc \
+ linux-headers \
+ build-base \
+ bash \
+ nodejs \
+ npm && break || sleep 5; \
+ done \
&& pip install --no-cache-dir --upgrade pip build
-# Copy project files
+# Cache Python dependencies
+COPY requirements.txt .
+RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt \
+ && pip wheel --no-cache-dir --wheel-dir=/wheels/ "semantic_router==0.1.11" "aurelio-sdk==0.0.19" "PyJWT==2.9.0"
+
+# Copy source after dependency layers
COPY . .
-# Set LITELLM_NON_ROOT flag for build time
+# Set non-root flag for build time consistency
ENV LITELLM_NON_ROOT=true
-# Build Admin UI
-RUN mkdir -p /tmp/litellm_ui
+# Build Admin UI using the upstream command order while keeping a single RUN layer
+RUN mkdir -p /tmp/litellm_ui && \
+ npm install -g npm@latest && npm cache clean --force && \
+ cd /app/ui/litellm-dashboard && \
+ if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
+ cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
+ fi && \
+ rm -f package-lock.json && \
+ npm install --legacy-peer-deps && \
+ npm run build && \
+ cp -r /app/ui/litellm-dashboard/out/* /tmp/litellm_ui/ && \
+ mkdir -p /tmp/litellm_assets && \
+ cp /app/litellm/proxy/logo.jpg /tmp/litellm_assets/logo.jpg && \
+ ( cd /tmp/litellm_ui && \
+ for html_file in *.html; do \
+ if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
+ folder_name="${html_file%.html}" && \
+ mkdir -p "$folder_name" && \
+ mv "$html_file" "$folder_name/index.html"; \
+ fi; \
+ done ) && \
+ cd /app/ui/litellm-dashboard && rm -rf ./out
-RUN npm install -g npm@latest && npm cache clean --force
-
-RUN cd /app/ui/litellm-dashboard && \
- if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
- cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
- fi
-
-RUN cd /app/ui/litellm-dashboard && rm -f package-lock.json
-
-RUN cd /app/ui/litellm-dashboard && npm install --legacy-peer-deps
-
-RUN cd /app/ui/litellm-dashboard && npm run build
-
-RUN cp -r /app/ui/litellm-dashboard/out/* /tmp/litellm_ui/
-RUN mkdir -p /tmp/litellm_assets && cp /app/litellm/proxy/logo.jpg /tmp/litellm_assets/logo.jpg
-
-RUN cd /tmp/litellm_ui && \
- for html_file in *.html; do \
- if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
- folder_name="${html_file%.html}" && \
- mkdir -p "$folder_name" && \
- mv "$html_file" "$folder_name/index.html"; \
- fi; \
- done
-
-RUN cd /app/ui/litellm-dashboard && rm -rf ./out
-
-# Build package and wheel dependencies
+# Build litellm wheel and place it in wheels dir (replace any PyPI wheels)
RUN rm -rf dist/* && python -m build && \
- pip install dist/*.whl && \
- pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
+ rm -f /wheels/litellm-*.whl && \
+ cp dist/*.whl /wheels/
+
+# Optionally build local litellm-proxy-extras wheel
+RUN if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
+ cd /app/litellm-proxy-extras && rm -rf dist && python -m build && \
+ cp dist/*.whl /wheels/; \
+ fi
+
+# Pre-cache Prisma binaries in the builder stage
+ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
+ PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
+ XDG_CACHE_HOME=/app/.cache \
+ PATH="/usr/lib/python3.13/site-packages/nodejs/bin:${PATH}"
+
+RUN pip install --no-cache-dir prisma==0.11.0 nodejs-bin==18.4.0a4 \
+ && mkdir -p /app/.cache/npm
+
+RUN NPM_CONFIG_CACHE=/app/.cache/npm \
+ python -c "import prisma.cli.prisma as p; p.ensure_cached()"
+
+RUN prisma generate && \
+ prisma --version && \
+ prisma migrate diff --from-empty --to-schema-datamodel ./schema.prisma --script > /dev/null 2>&1 || true
# -----------------
# Runtime Stage
# -----------------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
+ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
-
-# Install runtime dependencies
USER root
-RUN for i in 1 2 3; do \
- apk upgrade --no-cache && break || sleep 5; \
- done \
- && for i in 1 2 3; do \
- apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
- done
-# Copy only necessary artifacts from builder stage for runtime
-COPY . .
+# Install runtime dependencies with retry
+RUN for i in 1 2 3; do \
+ apk upgrade --no-cache && break || sleep 5; \
+ done \
+ && for i in 1 2 3; do \
+ apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
+ done
+
+# Copy artifacts from builder
+COPY --from=builder /app/requirements.txt /app/requirements.txt
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
-COPY --from=builder /app/schema.prisma /app/schema.prisma
-COPY --from=builder /app/dist/*.whl .
+COPY --from=builder /app/schema.prisma /app/
COPY --from=builder /wheels/ /wheels/
COPY --from=builder /tmp/litellm_ui /tmp/litellm_ui
COPY --from=builder /tmp/litellm_assets /tmp/litellm_assets
+COPY --from=builder /app/.cache /app/.cache
+COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
+COPY --from=builder \
+ /usr/lib/python3.13/site-packages/nodejs* \
+ /usr/lib/python3.13/site-packages/prisma* \
+ /usr/lib/python3.13/site-packages/tomlkit* \
+ /usr/lib/python3.13/site-packages/nodeenv* \
+ /usr/lib/python3.13/site-packages/
+COPY --from=builder /usr/bin/prisma /usr/bin/prisma
-# Install package from wheel and dependencies
-RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
- && rm -f *.whl \
- && rm -rf /wheels
+# Final runtime environment configuration
+ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
+ PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
+ HOME=/app \
+ LITELLM_NON_ROOT=true \
+ XDG_CACHE_HOME=/app/.cache
-# Remove test files and keys from dependencies
-RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
- find /usr/lib -type d -path "*/tornado/test" -delete
+# Install packages from wheels and optional extras without network
+RUN pip install --no-index --find-links=/wheels/ -r requirements.txt && \
+ pip install --no-index --find-links=/wheels/ /wheels/litellm-*-py3-none-any.whl && \
+ pip install --no-index --find-links=/wheels/ --no-deps semantic_router==0.1.11 && \
+ pip install --no-index --find-links=/wheels/ aurelio-sdk==0.0.19 && \
+ if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
+ if ls /wheels/litellm_proxy_extras-*.whl >/dev/null 2>&1; then \
+ pip install --no-index --find-links=/wheels/ /wheels/litellm_proxy_extras-*.whl; \
+ else \
+ echo "litellm_proxy_extras wheel not found; skipping local install"; \
+ fi; \
+ fi
-# Install semantic_router and aurelio-sdk using script
-RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
+# Permissions, cleanup, and Prisma prep
+RUN chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh && \
+ mkdir -p /nonexistent /.npm /tmp/litellm_assets /tmp/litellm_ui && \
+ chown -R nobody:nogroup /app /tmp/litellm_ui /tmp/litellm_assets /nonexistent /.npm && \
+ pip uninstall jwt -y || true && \
+ pip uninstall PyJWT -y || true && \
+ pip install --no-index --find-links=/wheels/ PyJWT==2.10.1 --no-cache-dir && \
+ rm -rf /wheels && \
+ PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
+ chown -R nobody:nogroup $PRISMA_PATH && \
+ LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
+ [ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH && \
+ LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
+ chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
+ [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
+ chmod -R g=u $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
+ [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
+ chmod -R g+w $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
+ [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true && \
+ chmod -R g+rX $PRISMA_PATH && \
+ chmod -R g+rX /app/.cache && \
+ mkdir -p /tmp/.npm /nonexistent /.npm && \
+ prisma generate
-# Ensure correct JWT library is used (pyjwt not jwt)
-RUN pip uninstall jwt -y && \
- pip uninstall PyJWT -y && \
- pip install PyJWT==2.9.0 --no-cache-dir
-
-# Set Prisma cache directories
-ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
-ENV NPM_CONFIG_CACHE=/.npm
-
-# Install prisma and make entrypoints executable
-RUN pip install --no-cache-dir prisma && \
- chmod +x docker/entrypoint.sh && \
- chmod +x docker/prod_entrypoint.sh
-
-# Create directories and set permissions for non-root user
-RUN mkdir -p /nonexistent /.npm /tmp/litellm_assets && \
- chown -R nobody:nogroup /app /tmp/litellm_ui /tmp/litellm_assets /nonexistent /.npm && \
- PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
- chown -R nobody:nogroup $PRISMA_PATH && \
- LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
- [ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH
-
-# OpenShift compatibility
-RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
- LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
- chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
- [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
- chmod -R g=u $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
- [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
- chmod -R g+w $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
- [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true
-
-# Switch to non-root user
+# Switch to non-root user for runtime
USER nobody
-# Set HOME for prisma generate to have a writable directory
-ENV HOME=/app
-
-# Set LITELLM_NON_ROOT flag for runtime
-ENV LITELLM_NON_ROOT=true
-
-RUN prisma generate
+# Prisma runtime knobs for offline containers
+ENV PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
+ PRISMA_HIDE_UPDATE_MESSAGE=1 \
+ PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
+ NPM_CONFIG_CACHE=/app/.cache/npm \
+ NPM_CONFIG_PREFER_OFFLINE=true \
+ PRISMA_OFFLINE_MODE=true
EXPOSE 4000/tcp
-
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
-
-CMD ["--port", "4000"]
\ No newline at end of file
+CMD ["--port", "4000"]
diff --git a/docker/README.md b/docker/README.md
index ce478dfe0dd..6d81276bb4b 100644
--- a/docker/README.md
+++ b/docker/README.md
@@ -59,6 +59,30 @@ To stop the running containers, use the following command:
docker compose down
```
+## Hardened / Offline Testing
+
+To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file:
+
+```bash
+docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache
+docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d
+```
+
+This setup:
+- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image.
+- Runs the proxy as a non-root user with a read-only rootfs and only two writable tmpfs mounts:
+ - `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`)
+ - `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`)
+- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines.
+
+You should also verify offline Prisma behaviour with:
+
+```bash
+docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version
+```
+
+This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access.
+
## Troubleshooting
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
diff --git a/docs/my-website/docs/a2a.md b/docs/my-website/docs/a2a.md
index 9c94a2fbf29..d7145e4b83c 100644
--- a/docs/my-website/docs/a2a.md
+++ b/docs/my-website/docs/a2a.md
@@ -16,7 +16,7 @@ Add A2A Agents on LiteLLM AI Gateway, Invoke agents in A2A Protocol, track reque
| Feature | Supported |
|---------|-----------|
-| Supported Agent Providers | A2A, LangGraph, Azure AI Foundry, Bedrock AgentCore |
+| Supported Agent Providers | A2A, Vertex AI Agent Engine, LangGraph, Azure AI Foundry, Bedrock AgentCore, Pydantic AI |
| Logging | ✅ |
| Load Balancing | ✅ |
| Streaming | ✅ |
@@ -45,17 +45,26 @@ You can add A2A-compatible agents through the LiteLLM Admin UI.
The URL should be the invocation URL for your A2A agent (e.g., `http://localhost:10001`).
+
### Add Azure AI Foundry Agents
Follow [this guide, to add your azure ai foundry agent to LiteLLM Agent Gateway](./providers/azure_ai_agents#litellm-a2a-gateway)
+### Add Vertex AI Agent Engine
+
+Follow [this guide, to add your Vertex AI Agent Engine to LiteLLM Agent Gateway](./providers/vertex_ai_agent_engine)
+
+### Add Bedrock AgentCore Agents
+
+Follow [this guide, to add your bedrock agentcore agent to LiteLLM Agent Gateway](./providers/bedrock_agentcore#litellm-a2a-gateway)
+
### Add LangGraph Agents
Follow [this guide, to add your langgraph agent to LiteLLM Agent Gateway](./providers/langgraph#litellm-a2a-gateway)
-### Add Bedrock AgentCore Agents
+### Add Pydantic AI Agents
-Follow [this guide, to add your bedrock agentcore agent to LiteLLM Agent Gateway](./providers/bedrock_agentcore#litellm-a2a-gateway)
+Follow [this guide, to add your pydantic ai agent to LiteLLM Agent Gateway](./providers/pydantic_ai_agent#litellm-a2a-gateway)
## Invoking your Agents
diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md
index f393b300f73..ba605e316d3 100644
--- a/docs/my-website/docs/index.md
+++ b/docs/my-website/docs/index.md
@@ -657,7 +657,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml --detailed_debug
```
diff --git a/docs/my-website/docs/observability/datadog.md b/docs/my-website/docs/observability/datadog.md
index b2901650ea6..7cf91ced34c 100644
--- a/docs/my-website/docs/observability/datadog.md
+++ b/docs/my-website/docs/observability/datadog.md
@@ -181,7 +181,7 @@ docker run \
-e USE_DDTRACE=true \
-e USE_DDPROFILER=true \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml --detailed_debug
```
diff --git a/docs/my-website/docs/providers/pydantic_ai_agent.md b/docs/my-website/docs/providers/pydantic_ai_agent.md
new file mode 100644
index 00000000000..e96295faaf3
--- /dev/null
+++ b/docs/my-website/docs/providers/pydantic_ai_agent.md
@@ -0,0 +1,121 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Pydantic AI Agents
+
+Call Pydantic AI Agents via LiteLLM's A2A Gateway.
+
+| Property | Details |
+|----------|---------|
+| Description | Pydantic AI agents with native A2A support via the `to_a2a()` method. LiteLLM provides fake streaming support for agents that don't natively stream. |
+| Provider Route on LiteLLM | A2A Gateway |
+| Supported Endpoints | `/v1/a2a/message/send` |
+| Provider Doc | [Pydantic AI Agents ↗](https://ai.pydantic.dev/agents/) |
+
+## LiteLLM A2A Gateway
+
+All Pydantic AI agents need to be exposed as A2A agents using the `to_a2a()` method. Once your agent server is running, you can add it to the LiteLLM Gateway.
+
+### 1. Setup Pydantic AI Agent Server
+
+LiteLLM requires Pydantic AI agents to follow the [A2A (Agent-to-Agent) protocol](https://github.com/google/A2A). Pydantic AI has native A2A support via the `to_a2a()` method, which exposes your agent as an A2A-compliant server.
+
+#### Install Dependencies
+
+```bash
+pip install pydantic-ai fasta2a uvicorn
+```
+
+#### Create Agent
+
+```python title="agent.py"
+from pydantic_ai import Agent
+
+agent = Agent('openai:gpt-4o-mini', instructions='Be helpful!')
+
+@agent.tool_plain
+def get_weather(city: str) -> str:
+ """Get weather for a city."""
+ return f"Weather in {city}: Sunny, 72°F"
+
+@agent.tool_plain
+def calculator(expression: str) -> str:
+ """Evaluate a math expression."""
+ return str(eval(expression))
+
+# Native A2A server - Pydantic AI handles it automatically
+app = agent.to_a2a()
+```
+
+#### Run Server
+
+```bash
+uvicorn agent:app --host 0.0.0.0 --port 9999
+```
+
+Server runs at `http://localhost:9999`
+
+### 2. Navigate to Agents
+
+From the sidebar, click "Agents" to open the agent management page, then click "+ Add New Agent".
+
+### 3. Select Pydantic AI Agent Type
+
+Click "A2A Standard" to see available agent types, then select "Pydantic AI".
+
+
+
+
+
+### 4. Configure the Agent
+
+Fill in the following fields:
+
+- **Agent Name** - A unique identifier for your agent (e.g., `test-pydantic-agent`)
+- **Agent URL** - The URL where your Pydantic AI agent is running. We use `http://localhost:9999` because that's where we started our Pydantic AI agent server in the previous step.
+
+
+
+
+
+
+
+### 5. Create Agent
+
+Click "Create Agent" to save your configuration.
+
+
+
+### 6. Test in Playground
+
+Go to "Playground" in the sidebar to test your agent.
+
+
+
+### 7. Select A2A Endpoint
+
+Click the endpoint dropdown and search for "a2a", then select `/v1/a2a/message/send`.
+
+
+
+
+
+
+
+### 8. Select Your Agent and Send a Message
+
+Pick your Pydantic AI agent from the dropdown and send a test message.
+
+
+
+
+
+
+
+
+## Further Reading
+
+- [Pydantic AI Documentation](https://ai.pydantic.dev/)
+- [Pydantic AI Agents](https://ai.pydantic.dev/agents/)
+- [A2A Agent Gateway](../a2a.md)
+- [A2A Cost Tracking](../a2a_cost_tracking.md)
diff --git a/docs/my-website/docs/providers/vertex_ai_agent_engine.md b/docs/my-website/docs/providers/vertex_ai_agent_engine.md
new file mode 100644
index 00000000000..3bd40e98684
--- /dev/null
+++ b/docs/my-website/docs/providers/vertex_ai_agent_engine.md
@@ -0,0 +1,216 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Vertex AI Agent Engine
+
+Call Vertex AI Agent Engine (Reasoning Engines) in the OpenAI Request/Response format.
+
+| Property | Details |
+|----------|---------|
+| Description | Vertex AI Agent Engine provides hosted agent runtimes that can execute agentic workflows with foundation models, tools, and custom logic. |
+| Provider Route on LiteLLM | `vertex_ai/agent_engine/{RESOURCE_NAME}` |
+| Supported Endpoints | `/chat/completions`, `/v1/messages`, `/v1/responses`, `/v1/a2a/message/send` |
+| Provider Doc | [Vertex AI Agent Engine ↗](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/overview) |
+
+## Quick Start
+
+### Model Format
+
+```shell showLineNumbers title="Model Format"
+vertex_ai/agent_engine/{RESOURCE_NAME}
+```
+
+**Example:**
+- `vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888`
+
+### LiteLLM Python SDK
+
+```python showLineNumbers title="Basic Agent Completion"
+import litellm
+
+response = litellm.completion(
+ model="vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888",
+ messages=[
+ {"role": "user", "content": "Explain machine learning in simple terms"}
+ ],
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Streaming Agent Responses"
+import litellm
+
+response = await litellm.acompletion(
+ model="vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888",
+ messages=[
+ {"role": "user", "content": "What are the key principles of software architecture?"}
+ ],
+ stream=True,
+)
+
+async for chunk in response:
+ if chunk.choices[0].delta.content:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+### LiteLLM Proxy
+
+#### 1. Configure your model in config.yaml
+
+
+
+
+```yaml showLineNumbers title="LiteLLM Proxy Configuration"
+model_list:
+ - model_name: vertex-agent-1
+ litellm_params:
+ model: vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888
+ vertex_project: your-project-id
+ vertex_location: us-central1
+```
+
+
+
+
+#### 2. Start the LiteLLM Proxy
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+```
+
+#### 3. Make requests to your Vertex AI Agent Engine
+
+
+
+
+```bash showLineNumbers title="Basic Agent Request"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -d '{
+ "model": "vertex-agent-1",
+ "messages": [
+ {"role": "user", "content": "Summarize the main benefits of cloud computing"}
+ ]
+ }'
+```
+
+
+
+
+
+```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
+from openai import OpenAI
+
+client = OpenAI(
+ base_url="http://localhost:4000",
+ api_key="your-litellm-api-key"
+)
+
+response = client.chat.completions.create(
+ model="vertex-agent-1",
+ messages=[
+ {"role": "user", "content": "What are best practices for API design?"}
+ ]
+)
+
+print(response.choices[0].message.content)
+```
+
+
+
+
+## LiteLLM A2A Gateway
+
+You can also connect to Vertex AI Agent Engine through LiteLLM's A2A (Agent-to-Agent) Gateway UI. This provides a visual way to register and test agents without writing code.
+
+### 1. Navigate to Agents
+
+From the sidebar, click "Agents" to open the agent management page, then click "+ Add New Agent".
+
+
+
+
+
+### 2. Select Vertex AI Agent Engine Type
+
+Click "A2A Standard" to see available agent types, then select "Vertex AI Agent Engine".
+
+
+
+
+
+### 3. Configure the Agent
+
+Fill in the following fields:
+
+- **Agent Name** - A friendly name for your agent (e.g., `my-vertex-agent`)
+- **Reasoning Engine Resource ID** - The full resource path from Google Cloud Console (e.g., `projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888`)
+- **Vertex Project** - Your Google Cloud project ID
+- **Vertex Location** - The region where your agent is deployed (e.g., `us-central1`)
+
+
+
+
+
+You can find the Resource ID in Google Cloud Console under Vertex AI > Agent Engine:
+
+
+
+
+
+You can find the Project ID in Google Cloud Console:
+
+
+
+
+
+### 4. Create Agent
+
+Click "Create Agent" to save your configuration.
+
+
+
+### 5. Test in Playground
+
+Go to "Playground" in the sidebar to test your agent.
+
+
+
+### 6. Select A2A Endpoint
+
+Click the endpoint dropdown and select `/v1/a2a/message/send`.
+
+
+
+### 7. Select Your Agent and Send a Message
+
+Pick your Vertex AI Agent Engine from the dropdown and send a test message.
+
+
+
+
+
+
+
+## Environment Variables
+
+| Variable | Description |
+|----------|-------------|
+| `GOOGLE_APPLICATION_CREDENTIALS` | Path to service account JSON key file |
+| `VERTEXAI_PROJECT` | Google Cloud project ID |
+| `VERTEXAI_LOCATION` | Google Cloud region (default: `us-central1`) |
+
+```bash
+export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service-account.json"
+export VERTEXAI_PROJECT="your-project-id"
+export VERTEXAI_LOCATION="us-central1"
+```
+
+## Further Reading
+
+- [Vertex AI Agent Engine Documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/overview)
+- [Create a Reasoning Engine](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/create)
+- [A2A Agent Gateway](../a2a.md)
+- [Vertex AI Provider](./vertex.md)
diff --git a/docs/my-website/docs/proxy/configs.md b/docs/my-website/docs/proxy/configs.md
index 77ab3158f74..ba4ca190aa9 100644
--- a/docs/my-website/docs/proxy/configs.md
+++ b/docs/my-website/docs/proxy/configs.md
@@ -655,7 +655,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-e LITELLM_CONFIG_BUCKET_TYPE="gcs" \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest --detailed_debug
+ docker.litellm.ai/berriai/litellm-database:main-latest --detailed_debug
```
@@ -676,7 +676,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_NAME= \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest
+ docker.litellm.ai/berriai/litellm-database:main-latest
```
diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md
index 0f0e5f678d3..9b4bc6822c1 100644
--- a/docs/my-website/docs/proxy/deploy.md
+++ b/docs/my-website/docs/proxy/deploy.md
@@ -10,10 +10,38 @@ You can find the Dockerfile to build litellm proxy [here](https://github.com/Ber
## Quick Start
+:::info
+Facing issues with pulling the docker image? Email us at support@berri.ai.
+:::
+
To start using Litellm, run the following commands in a shell:
+
+
+
+
+```
+docker pull docker.litellm.ai/berriai/litellm:main-latest
+```
+
+[**See all docker images**](https://github.com/orgs/BerriAI/packages)
+
+
+
+
+
+```shell
+$ pip install 'litellm[proxy]'
+```
+
+
+
+
+
+Use this docker compose to spin up the proxy with a postgres database running locally.
+
```bash
-# Get the code
+# Get the docker compose file
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/docker-compose.yml
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/prometheus.yml
@@ -30,6 +58,8 @@ echo 'LITELLM_SALT_KEY="sk-1234"' >> .env
docker compose up
```
+
+
### Docker Run
@@ -57,7 +87,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-stable \
+ docker.litellm.ai/berriai/litellm:main-stable \
--config /app/config.yaml --detailed_debug
```
@@ -87,12 +117,12 @@ See all supported CLI args [here](https://docs.litellm.ai/docs/proxy/cli):
Here's how you can run the docker image and pass your config to `litellm`
```shell
-docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
+docker run docker.litellm.ai/berriai/litellm:main-stable --config your_config.yaml
```
Here's how you can run the docker image and start litellm on port 8002 with `num_workers=8`
```shell
-docker run ghcr.io/berriai/litellm:main-stable --port 8002 --num_workers 8
+docker run docker.litellm.ai/berriai/litellm:main-stable --port 8002 --num_workers 8
```
@@ -100,7 +130,7 @@ docker run ghcr.io/berriai/litellm:main-stable --port 8002 --num_workers 8
```shell
# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-stable
+FROM docker.litellm.ai/berriai/litellm:main-stable
# Set the working directory to /app
WORKDIR /app
@@ -242,7 +272,7 @@ spec:
spec:
containers:
- name: litellm
- image: ghcr.io/berriai/litellm:main-stable # it is recommended to fix a version generally
+ image: docker.litellm.ai/berriai/litellm:main-stable # it is recommended to fix a version generally
args:
- "--config"
- "/app/proxy_server_config.yaml"
@@ -279,9 +309,9 @@ Use this when you want to use litellm helm chart as a dependency for other chart
#### Step 1. Pull the litellm helm chart
```bash
-helm pull oci://ghcr.io/berriai/litellm-helm
+helm pull oci://docker.litellm.ai/berriai/litellm-helm
-# Pulled: ghcr.io/berriai/litellm-helm:0.1.2
+# Pulled: docker.litellm.ai/berriai/litellm-helm:0.1.2
# Digest: sha256:7d3ded1c99c1597f9ad4dc49d84327cf1db6e0faa0eeea0c614be5526ae94e2a
```
@@ -340,7 +370,7 @@ Requirements:
We maintain a [separate Dockerfile](https://github.com/BerriAI/litellm/pkgs/container/litellm-database) for reducing build time when running LiteLLM proxy with a connected Postgres Database
```shell
-docker pull ghcr.io/berriai/litellm-database:main-stable
+docker pull docker.litellm.ai/berriai/litellm-database:main-stable
```
```shell
@@ -351,7 +381,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-stable \
+ docker.litellm.ai/berriai/litellm-database:main-stable \
--config /app/config.yaml --detailed_debug
```
@@ -379,7 +409,7 @@ spec:
spec:
containers:
- name: litellm-container
- image: ghcr.io/berriai/litellm:main-stable
+ image: docker.litellm.ai/berriai/litellm:main-stable
imagePullPolicy: Always
env:
- name: AZURE_API_KEY
@@ -516,9 +546,9 @@ Use this when you want to use litellm helm chart as a dependency for other chart
#### Step 1. Pull the litellm helm chart
```bash
-helm pull oci://ghcr.io/berriai/litellm-helm
+helm pull oci://docker.litellm.ai/berriai/litellm-helm
-# Pulled: ghcr.io/berriai/litellm-helm:0.1.2
+# Pulled: docker.litellm.ai/berriai/litellm-helm:0.1.2
# Digest: sha256:7d3ded1c99c1597f9ad4dc49d84327cf1db6e0faa0eeea0c614be5526ae94e2a
```
@@ -575,7 +605,7 @@ router_settings:
Start docker container with config
```shell
-docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
+docker run docker.litellm.ai/berriai/litellm:main-stable --config your_config.yaml
```
### Deploy with Database + Redis
@@ -610,7 +640,7 @@ Start `litellm-database`docker container with config
docker run --name litellm-proxy \
-e DATABASE_URL=postgresql://:@:/ \
-p 4000:4000 \
-ghcr.io/berriai/litellm-database:main-stable --config your_config.yaml
+docker.litellm.ai/berriai/litellm-database:main-stable --config your_config.yaml
```
### (Non Root) - without Internet Connection
@@ -620,7 +650,7 @@ By default `prisma generate` downloads [prisma's engine binaries](https://www.pr
Use this docker image to deploy litellm with pre-generated prisma binaries.
```bash
-docker pull ghcr.io/berriai/litellm-non_root:main-stable
+docker pull docker.litellm.ai/berriai/litellm-non_root:main-stable
```
[Published Docker Image link](https://github.com/BerriAI/litellm/pkgs/container/litellm-non_root)
@@ -639,7 +669,7 @@ Use this, If you need to set ssl certificates for your on prem litellm proxy
Pass `ssl_keyfile_path` (Path to the SSL keyfile) and `ssl_certfile_path` (Path to the SSL certfile) when starting litellm proxy
```shell
-docker run ghcr.io/berriai/litellm:main-stable \
+docker run docker.litellm.ai/berriai/litellm:main-stable \
--ssl_keyfile_path ssl_test/keyfile.key \
--ssl_certfile_path ssl_test/certfile.crt
```
@@ -654,7 +684,7 @@ Step 1. Build your custom docker image with hypercorn
```shell
# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-stable
+FROM docker.litellm.ai/berriai/litellm:main-stable
# Set the working directory to /app
WORKDIR /app
@@ -702,7 +732,7 @@ Usage Example:
In this example, we set the keepalive timeout to 75 seconds.
```shell showLineNumbers title="docker run"
-docker run ghcr.io/berriai/litellm:main-stable \
+docker run docker.litellm.ai/berriai/litellm:main-stable \
--keepalive_timeout 75
```
@@ -711,7 +741,7 @@ In this example, we set the keepalive timeout to 75 seconds.
```shell showLineNumbers title="Environment Variable"
export KEEPALIVE_TIMEOUT=75
-docker run ghcr.io/berriai/litellm:main-stable
+docker run docker.litellm.ai/berriai/litellm:main-stable
```
@@ -722,7 +752,7 @@ Use this to mitigate memory growth by recycling workers after a fixed number of
Usage Examples:
```shell showLineNumbers title="docker run (CLI flag)"
-docker run ghcr.io/berriai/litellm:main-stable \
+docker run docker.litellm.ai/berriai/litellm:main-stable \
--max_requests_before_restart 10000
```
@@ -730,7 +760,7 @@ Or set via environment variable:
```shell showLineNumbers title="Environment Variable"
export MAX_REQUESTS_BEFORE_RESTART=10000
-docker run ghcr.io/berriai/litellm:main-stable
+docker run docker.litellm.ai/berriai/litellm:main-stable
```
@@ -759,7 +789,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-e LITELLM_CONFIG_BUCKET_TYPE="gcs" \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-stable --detailed_debug
+ docker.litellm.ai/berriai/litellm-database:main-stable --detailed_debug
```
@@ -780,7 +810,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_NAME= \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-stable
+ docker.litellm.ai/berriai/litellm-database:main-stable
```
@@ -907,7 +937,7 @@ Run the following command, replacing `` with the value you copied
docker run --name litellm-proxy \
-e DATABASE_URL= \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-stable
+ docker.litellm.ai/berriai/litellm-database:main-stable
```
#### 4. Access the Application:
@@ -986,7 +1016,7 @@ services:
context: .
args:
target: runtime
- image: ghcr.io/berriai/litellm:main-stable
+ image: docker.litellm.ai/berriai/litellm:main-stable
ports:
- "4000:4000" # Map the container port to the host, change the host port if necessary
volumes:
diff --git a/docs/my-website/docs/proxy/docker_quick_start.md b/docs/my-website/docs/proxy/docker_quick_start.md
index 35d9923e92c..efdc73de43e 100644
--- a/docs/my-website/docs/proxy/docker_quick_start.md
+++ b/docs/my-website/docs/proxy/docker_quick_start.md
@@ -20,7 +20,7 @@ End-to-End tutorial for LiteLLM Proxy to:
```
-docker pull ghcr.io/berriai/litellm:main-latest
+docker pull docker.litellm.ai/berriai/litellm:main-latest
```
[**See all docker images**](https://github.com/orgs/BerriAI/packages)
@@ -119,7 +119,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml --detailed_debug
# RUNNING on http://0.0.0.0:4000
@@ -302,7 +302,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml --detailed_debug
```
diff --git a/docs/my-website/docs/proxy/guardrails/pangea.md b/docs/my-website/docs/proxy/guardrails/pangea.md
index 180b9100d6b..3de5ddfa530 100644
--- a/docs/my-website/docs/proxy/guardrails/pangea.md
+++ b/docs/my-website/docs/proxy/guardrails/pangea.md
@@ -67,7 +67,7 @@ docker run --rm \
-e PANGEA_AI_GUARD_TOKEN=$PANGEA_AI_GUARD_TOKEN \
-e OPENAI_API_KEY=$OPENAI_API_KEY \
-v $(pwd)/config.yaml:/app/config.yaml \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml
```
diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md
index 54c917bbbca..4cff7e5d041 100644
--- a/docs/my-website/docs/proxy/load_balancing.md
+++ b/docs/my-website/docs/proxy/load_balancing.md
@@ -29,6 +29,10 @@ LiteLLM automatically distributes requests across multiple deployments of the sa
| **latency-based-routing** | Routes to fastest responding deployment | Latency-critical applications |
| **cost-based-routing** | Routes to deployment with lowest cost | Cost-sensitive applications |
+:::tip Deployment Priority
+Use the `order` parameter to prioritize specific deployments. [See Deployment Ordering](#deployment-ordering-priority) for details.
+:::
+
## Quick Start - Load Balancing
#### Step 1 - Set deployments on config
@@ -243,6 +247,27 @@ class RouterModelGroupAliasItem(TypedDict):
hidden: bool # if 'True', don't return on `/v1/models`, `/v1/model/info`, `/v1/model_group/info`
```
+## Deployment Ordering (Priority)
+
+Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them.
+
+```yaml
+model_list:
+ - model_name: gpt-4
+ litellm_params:
+ model: azure/gpt-4-primary
+ api_key: os.environ/AZURE_API_KEY
+ order: 1 # 👈 Highest priority - always tried first
+
+ - model_name: gpt-4
+ litellm_params:
+ model: azure/gpt-4-fallback
+ api_key: os.environ/AZURE_API_KEY_2
+ order: 2 # 👈 Used when order=1 is unavailable
+```
+
+If `order=1` deployment is unavailable (e.g., rate-limited), the router falls back to `order=2` deployments.
+
### When You'll See Load Balancing in Action
**Immediate Effects:**
diff --git a/docs/my-website/docs/proxy/shared_health_check.md b/docs/my-website/docs/proxy/shared_health_check.md
index d4b70116309..c9c975c7911 100644
--- a/docs/my-website/docs/proxy/shared_health_check.md
+++ b/docs/my-website/docs/proxy/shared_health_check.md
@@ -269,7 +269,7 @@ spec:
spec:
containers:
- name: litellm-proxy
- image: ghcr.io/berriai/litellm:latest
+ image: docker.litellm.ai/berriai/litellm:latest
env:
- name: USE_SHARED_HEALTH_CHECK
value: "true"
diff --git a/docs/my-website/docs/routing.md b/docs/my-website/docs/routing.md
index 971427806ed..2539f70d5bc 100644
--- a/docs/my-website/docs/routing.md
+++ b/docs/my-website/docs/routing.md
@@ -832,6 +832,59 @@ asyncio.run(router_acompletion())
## Basic Reliability
+### Deployment Ordering (Priority)
+
+Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them.
+
+
+
+
+```python
+from litellm import Router
+
+model_list = [
+ {
+ "model_name": "gpt-4",
+ "litellm_params": {
+ "model": "azure/gpt-4-primary",
+ "api_key": os.getenv("AZURE_API_KEY"),
+ "order": 1, # 👈 Highest priority
+ },
+ },
+ {
+ "model_name": "gpt-4",
+ "litellm_params": {
+ "model": "azure/gpt-4-fallback",
+ "api_key": os.getenv("AZURE_API_KEY_2"),
+ "order": 2, # 👈 Used when order=1 is unavailable
+ },
+ },
+]
+
+router = Router(model_list=model_list)
+```
+
+
+
+
+```yaml
+model_list:
+ - model_name: gpt-4
+ litellm_params:
+ model: azure/gpt-4-primary
+ api_key: os.environ/AZURE_API_KEY
+ order: 1 # 👈 Highest priority
+
+ - model_name: gpt-4
+ litellm_params:
+ model: azure/gpt-4-fallback
+ api_key: os.environ/AZURE_API_KEY_2
+ order: 2 # 👈 Used when order=1 is unavailable
+```
+
+
+
+
### Weighted Deployments
Set `weight` on a deployment to pick one deployment more often than others.
diff --git a/docs/my-website/docs/secret_managers/custom_secret_manager.md b/docs/my-website/docs/secret_managers/custom_secret_manager.md
index c51eeeb0727..a6a91a0336d 100644
--- a/docs/my-website/docs/secret_managers/custom_secret_manager.md
+++ b/docs/my-website/docs/secret_managers/custom_secret_manager.md
@@ -76,7 +76,7 @@ docker run -d \
--name litellm-proxy \
-v $(pwd)/config.yaml:/app/config.yaml \
-v $(pwd)/my_secret_manager.py:/app/my_secret_manager.py \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml \
--port 4000 \
--detailed_debug
diff --git a/docs/my-website/docs/tutorials/elasticsearch_logging.md b/docs/my-website/docs/tutorials/elasticsearch_logging.md
index eabd47f095d..85a9f1452d7 100644
--- a/docs/my-website/docs/tutorials/elasticsearch_logging.md
+++ b/docs/my-website/docs/tutorials/elasticsearch_logging.md
@@ -221,7 +221,7 @@ services:
- elasticsearch
litellm:
- image: ghcr.io/berriai/litellm:main-latest
+ image: docker.litellm.ai/berriai/litellm:main-latest
ports:
- "4000:4000"
environment:
diff --git a/docs/my-website/docs/tutorials/openai_codex.md b/docs/my-website/docs/tutorials/openai_codex.md
index 41416f85159..563d6559ca5 100644
--- a/docs/my-website/docs/tutorials/openai_codex.md
+++ b/docs/my-website/docs/tutorials/openai_codex.md
@@ -53,7 +53,7 @@ yarn global add @openai/codex
docker run \
-v $(pwd)/litellm_config.yaml:/app/config.yaml \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml
```
diff --git a/docs/my-website/release_notes/v1.55.8-stable/index.md b/docs/my-website/release_notes/v1.55.8-stable/index.md
index 38c78eb5372..bf239e0889d 100644
--- a/docs/my-website/release_notes/v1.55.8-stable/index.md
+++ b/docs/my-website/release_notes/v1.55.8-stable/index.md
@@ -53,7 +53,7 @@ Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable
+docker.litellm.ai/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable
```
## Get Daily Updates
diff --git a/docs/my-website/release_notes/v1.57.3/index.md b/docs/my-website/release_notes/v1.57.3/index.md
index ab1154a0a8c..bbffa990b32 100644
--- a/docs/my-website/release_notes/v1.57.3/index.md
+++ b/docs/my-website/release_notes/v1.57.3/index.md
@@ -39,7 +39,7 @@ Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-
**You are only impacted if you use `apt-get` in your Dockerfile**
```shell
# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-latest
+FROM docker.litellm.ai/berriai/litellm:main-latest
# Set the working directory
WORKDIR /app
diff --git a/docs/my-website/release_notes/v1.63.11-stable/index.md b/docs/my-website/release_notes/v1.63.11-stable/index.md
index 882747a07b3..3273f9a8e06 100644
--- a/docs/my-website/release_notes/v1.63.11-stable/index.md
+++ b/docs/my-website/release_notes/v1.63.11-stable/index.md
@@ -36,7 +36,7 @@ This release is primarily focused on:
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.63.11-stable
+docker.litellm.ai/berriai/litellm:main-v1.63.11-stable
```
## Demo Instance
diff --git a/docs/my-website/release_notes/v1.63.14/index.md b/docs/my-website/release_notes/v1.63.14/index.md
index ff2630468c5..1ac713fc2d5 100644
--- a/docs/my-website/release_notes/v1.63.14/index.md
+++ b/docs/my-website/release_notes/v1.63.14/index.md
@@ -32,7 +32,7 @@ This release brings:
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1
+docker.litellm.ai/berriai/litellm:main-v1.63.14-stable.patch1
```
## Demo Instance
diff --git a/docs/my-website/release_notes/v1.65.4-stable/index.md b/docs/my-website/release_notes/v1.65.4-stable/index.md
index 872024a47ab..80d703e1116 100644
--- a/docs/my-website/release_notes/v1.65.4-stable/index.md
+++ b/docs/my-website/release_notes/v1.65.4-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.65.4-stable
+docker.litellm.ai/berriai/litellm:main-v1.65.4-stable
```
diff --git a/docs/my-website/release_notes/v1.66.0-stable/index.md b/docs/my-website/release_notes/v1.66.0-stable/index.md
index 939322e0317..693cd7fc5ac 100644
--- a/docs/my-website/release_notes/v1.66.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.66.0-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.66.0-stable
+docker.litellm.ai/berriai/litellm:main-v1.66.0-stable
```
diff --git a/docs/my-website/release_notes/v1.67.4-stable/index.md b/docs/my-website/release_notes/v1.67.4-stable/index.md
index 93a27155d2b..f61c99f7d02 100644
--- a/docs/my-website/release_notes/v1.67.4-stable/index.md
+++ b/docs/my-website/release_notes/v1.67.4-stable/index.md
@@ -30,7 +30,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.67.4-stable
+docker.litellm.ai/berriai/litellm:main-v1.67.4-stable
```
diff --git a/docs/my-website/release_notes/v1.68.0-stable/index.md b/docs/my-website/release_notes/v1.68.0-stable/index.md
index 4d456d9c853..f3e7fa27427 100644
--- a/docs/my-website/release_notes/v1.68.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.68.0-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.68.0-stable
+docker.litellm.ai/berriai/litellm:main-v1.68.0-stable
```
diff --git a/docs/my-website/release_notes/v1.69.0-stable/index.md b/docs/my-website/release_notes/v1.69.0-stable/index.md
index 3f8ce7a29c4..f3f094e5403 100644
--- a/docs/my-website/release_notes/v1.69.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.69.0-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.69.0-stable
+docker.litellm.ai/berriai/litellm:main-v1.69.0-stable
```
diff --git a/docs/my-website/release_notes/v1.70.1-stable/index.md b/docs/my-website/release_notes/v1.70.1-stable/index.md
index c55ac8b9c61..5d4bde0f6a0 100644
--- a/docs/my-website/release_notes/v1.70.1-stable/index.md
+++ b/docs/my-website/release_notes/v1.70.1-stable/index.md
@@ -30,7 +30,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.70.1-stable
+docker.litellm.ai/berriai/litellm:main-v1.70.1-stable
```
diff --git a/docs/my-website/release_notes/v1.71.1-stable/index.md b/docs/my-website/release_notes/v1.71.1-stable/index.md
index 2d21d49171b..bd37183455d 100644
--- a/docs/my-website/release_notes/v1.71.1-stable/index.md
+++ b/docs/my-website/release_notes/v1.71.1-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.71.1-stable
+docker.litellm.ai/berriai/litellm:main-v1.71.1-stable
```
diff --git a/docs/my-website/release_notes/v1.72.0-stable/index.md b/docs/my-website/release_notes/v1.72.0-stable/index.md
index 47bc19e8aa8..fe235cf07b1 100644
--- a/docs/my-website/release_notes/v1.72.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.72.0-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.72.0-stable
+docker.litellm.ai/berriai/litellm:main-v1.72.0-stable
```
diff --git a/docs/my-website/release_notes/v1.72.2-stable/index.md b/docs/my-website/release_notes/v1.72.2-stable/index.md
index 023180f9758..36d01c131c7 100644
--- a/docs/my-website/release_notes/v1.72.2-stable/index.md
+++ b/docs/my-website/release_notes/v1.72.2-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.72.2-stable
+docker.litellm.ai/berriai/litellm:main-v1.72.2-stable
```
diff --git a/docs/my-website/release_notes/v1.72.6-stable/index.md b/docs/my-website/release_notes/v1.72.6-stable/index.md
index 5603548364f..a20488e2318 100644
--- a/docs/my-website/release_notes/v1.72.6-stable/index.md
+++ b/docs/my-website/release_notes/v1.72.6-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run
-e STORE_MODEL_IN_DB=True
-p 4000:4000
-ghcr.io/berriai/litellm:main-v1.72.6-stable
+docker.litellm.ai/berriai/litellm:main-v1.72.6-stable
```
diff --git a/docs/my-website/release_notes/v1.73.0-stable/index.md b/docs/my-website/release_notes/v1.73.0-stable/index.md
index 307fecc36dd..802c5ac028b 100644
--- a/docs/my-website/release_notes/v1.73.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.73.0-stable/index.md
@@ -37,7 +37,7 @@ The `non-root` docker image has a known issue around the UI not loading. If you
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.73.0-stable
+docker.litellm.ai/berriai/litellm:v1.73.0-stable
```
diff --git a/docs/my-website/release_notes/v1.73.6-stable/index.md b/docs/my-website/release_notes/v1.73.6-stable/index.md
index b03380f9b2b..da748c5c99f 100644
--- a/docs/my-website/release_notes/v1.73.6-stable/index.md
+++ b/docs/my-website/release_notes/v1.73.6-stable/index.md
@@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.73.6-stable.patch.1
+docker.litellm.ai/berriai/litellm:v1.73.6-stable.patch.1
```
diff --git a/docs/my-website/release_notes/v1.74.0-stable/index.md b/docs/my-website/release_notes/v1.74.0-stable/index.md
index e49c2b4f620..ee39c0a26a8 100644
--- a/docs/my-website/release_notes/v1.74.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.74.0-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.74.0-stable
+docker.litellm.ai/berriai/litellm:v1.74.0-stable
```
diff --git a/docs/my-website/release_notes/v1.74.15-stable/index.md b/docs/my-website/release_notes/v1.74.15-stable/index.md
index 9807a00b7e7..c0facf8afb0 100644
--- a/docs/my-website/release_notes/v1.74.15-stable/index.md
+++ b/docs/my-website/release_notes/v1.74.15-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.74.15-stable
+docker.litellm.ai/berriai/litellm:v1.74.15-stable
```
diff --git a/docs/my-website/release_notes/v1.74.3-stable/index.md b/docs/my-website/release_notes/v1.74.3-stable/index.md
index 167d81e52af..05386172e71 100644
--- a/docs/my-website/release_notes/v1.74.3-stable/index.md
+++ b/docs/my-website/release_notes/v1.74.3-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.74.3-stable
+docker.litellm.ai/berriai/litellm:v1.74.3-stable
```
diff --git a/docs/my-website/release_notes/v1.74.7/index.md b/docs/my-website/release_notes/v1.74.7/index.md
index 7d7a568e13f..10fbd21b498 100644
--- a/docs/my-website/release_notes/v1.74.7/index.md
+++ b/docs/my-website/release_notes/v1.74.7/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.74.7-stable.patch.1
+docker.litellm.ai/berriai/litellm:v1.74.7-stable.patch.1
```
diff --git a/docs/my-website/release_notes/v1.74.9-stable/index.md b/docs/my-website/release_notes/v1.74.9-stable/index.md
index 3f100745dfe..9feed6d62e6 100644
--- a/docs/my-website/release_notes/v1.74.9-stable/index.md
+++ b/docs/my-website/release_notes/v1.74.9-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.74.9-stable.patch.1
+docker.litellm.ai/berriai/litellm:v1.74.9-stable.patch.1
```
diff --git a/docs/my-website/release_notes/v1.75.5-stable/index.md b/docs/my-website/release_notes/v1.75.5-stable/index.md
index 7035d285057..043f1267fc8 100644
--- a/docs/my-website/release_notes/v1.75.5-stable/index.md
+++ b/docs/my-website/release_notes/v1.75.5-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.75.5-stable
+docker.litellm.ai/berriai/litellm:v1.75.5-stable
```
diff --git a/docs/my-website/release_notes/v1.75.8/index.md b/docs/my-website/release_notes/v1.75.8/index.md
index d7d4f37c4ee..3db1fe4b2cd 100644
--- a/docs/my-website/release_notes/v1.75.8/index.md
+++ b/docs/my-website/release_notes/v1.75.8/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.75.8-stable
+docker.litellm.ai/berriai/litellm:v1.75.8-stable
```
diff --git a/docs/my-website/release_notes/v1.76.1-stable/index.md b/docs/my-website/release_notes/v1.76.1-stable/index.md
index 4437b7f5799..f458dfde6d4 100644
--- a/docs/my-website/release_notes/v1.76.1-stable/index.md
+++ b/docs/my-website/release_notes/v1.76.1-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.76.1
+docker.litellm.ai/berriai/litellm:v1.76.1
```
diff --git a/docs/my-website/release_notes/v1.76.3-stable/index.md b/docs/my-website/release_notes/v1.76.3-stable/index.md
index 6b40e4f5b35..9763a57975b 100644
--- a/docs/my-website/release_notes/v1.76.3-stable/index.md
+++ b/docs/my-website/release_notes/v1.76.3-stable/index.md
@@ -35,7 +35,7 @@ This release has a known issue where startup is leading to Out of Memory errors
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.76.3
+docker.litellm.ai/berriai/litellm:v1.76.3
```
diff --git a/docs/my-website/release_notes/v1.77.2-stable/index.md b/docs/my-website/release_notes/v1.77.2-stable/index.md
index fdd80693d05..4f732a1604d 100644
--- a/docs/my-website/release_notes/v1.77.2-stable/index.md
+++ b/docs/my-website/release_notes/v1.77.2-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:main-v1.77.2-stable
+docker.litellm.ai/berriai/litellm:main-v1.77.2-stable
```
diff --git a/docs/my-website/release_notes/v1.77.3-stable/index.md b/docs/my-website/release_notes/v1.77.3-stable/index.md
index c7c17e5baee..11b82c4c834 100644
--- a/docs/my-website/release_notes/v1.77.3-stable/index.md
+++ b/docs/my-website/release_notes/v1.77.3-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.77.3-stable
+docker.litellm.ai/berriai/litellm:v1.77.3-stable
```
diff --git a/docs/my-website/release_notes/v1.77.5-stable/index.md b/docs/my-website/release_notes/v1.77.5-stable/index.md
index 6843800ee6d..8e59ea92cc2 100644
--- a/docs/my-website/release_notes/v1.77.5-stable/index.md
+++ b/docs/my-website/release_notes/v1.77.5-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.77.5-stable
+docker.litellm.ai/berriai/litellm:v1.77.5-stable
```
diff --git a/docs/my-website/release_notes/v1.77.7-stable/index.md b/docs/my-website/release_notes/v1.77.7-stable/index.md
index 62d9a2eee4f..b4df447f334 100644
--- a/docs/my-website/release_notes/v1.77.7-stable/index.md
+++ b/docs/my-website/release_notes/v1.77.7-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.77.7.rc.1
+docker.litellm.ai/berriai/litellm:v1.77.7.rc.1
```
diff --git a/docs/my-website/release_notes/v1.78.0-stable/index.md b/docs/my-website/release_notes/v1.78.0-stable/index.md
index 7f6c5ba1e08..8322f0479c5 100644
--- a/docs/my-website/release_notes/v1.78.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.78.0-stable/index.md
@@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.78.0-stable
+docker.litellm.ai/berriai/litellm:v1.78.0-stable
```
diff --git a/docs/my-website/release_notes/v1.78.5-stable/index.md b/docs/my-website/release_notes/v1.78.5-stable/index.md
index af1fd359fa2..2bcdfab472c 100644
--- a/docs/my-website/release_notes/v1.78.5-stable/index.md
+++ b/docs/my-website/release_notes/v1.78.5-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.78.5-stable
+docker.litellm.ai/berriai/litellm:v1.78.5-stable
```
diff --git a/docs/my-website/release_notes/v1.79.0-stable/index.md b/docs/my-website/release_notes/v1.79.0-stable/index.md
index 8327f4b6178..4bb7094a3fc 100644
--- a/docs/my-website/release_notes/v1.79.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.79.0-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.79.0-stable
+docker.litellm.ai/berriai/litellm:v1.79.0-stable
```
diff --git a/docs/my-website/release_notes/v1.79.1-stable/index.md b/docs/my-website/release_notes/v1.79.1-stable/index.md
index ea8cfeae740..19fc7f9f3ff 100644
--- a/docs/my-website/release_notes/v1.79.1-stable/index.md
+++ b/docs/my-website/release_notes/v1.79.1-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.79.1-stable
+docker.litellm.ai/berriai/litellm:v1.79.1-stable
```
diff --git a/docs/my-website/release_notes/v1.79.3-stable/index.md b/docs/my-website/release_notes/v1.79.3-stable/index.md
index c4f3ba1e017..542f88787e0 100644
--- a/docs/my-website/release_notes/v1.79.3-stable/index.md
+++ b/docs/my-website/release_notes/v1.79.3-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.79.3-stable
+docker.litellm.ai/berriai/litellm:v1.79.3-stable
```
diff --git a/docs/my-website/release_notes/v1.80.0-stable/index.md b/docs/my-website/release_notes/v1.80.0-stable/index.md
index 17fcf6646ed..d0cf28a5c58 100644
--- a/docs/my-website/release_notes/v1.80.0-stable/index.md
+++ b/docs/my-website/release_notes/v1.80.0-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.80.0-stable
+docker.litellm.ai/berriai/litellm:v1.80.0-stable
```
diff --git a/docs/my-website/release_notes/v1.80.10-stable/index.md b/docs/my-website/release_notes/v1.80.10-stable/index.md
index 1b0a9866fae..2290c06de53 100644
--- a/docs/my-website/release_notes/v1.80.10-stable/index.md
+++ b/docs/my-website/release_notes/v1.80.10-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.80.10.rc.1
+docker.litellm.ai/berriai/litellm:v1.80.10.rc.1
```
diff --git a/docs/my-website/release_notes/v1.80.5-stable/index.md b/docs/my-website/release_notes/v1.80.5-stable/index.md
index 598fa47f223..9c769f8996f 100644
--- a/docs/my-website/release_notes/v1.80.5-stable/index.md
+++ b/docs/my-website/release_notes/v1.80.5-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.80.5-stable
+docker.litellm.ai/berriai/litellm:v1.80.5-stable
```
diff --git a/docs/my-website/release_notes/v1.80.8-stable/index.md b/docs/my-website/release_notes/v1.80.8-stable/index.md
index 29075a9594f..106c594968f 100644
--- a/docs/my-website/release_notes/v1.80.8-stable/index.md
+++ b/docs/my-website/release_notes/v1.80.8-stable/index.md
@@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
-ghcr.io/berriai/litellm:v1.80.8-stable
+docker.litellm.ai/berriai/litellm:v1.80.8-stable
```
diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js
index 954a27d3182..c64ac1e30fe 100644
--- a/docs/my-website/sidebars.js
+++ b/docs/my-website/sidebars.js
@@ -632,6 +632,7 @@ const sidebars = {
"providers/vertex_speech",
"providers/vertex_batch",
"providers/vertex_ocr",
+ "providers/vertex_ai_agent_engine",
]
},
{
@@ -738,6 +739,7 @@ const sidebars = {
"providers/petals",
"providers/publicai",
"providers/predibase",
+ "providers/pydantic_ai_agent",
"providers/ragflow",
"providers/recraft",
"providers/replicate",
diff --git a/docs/my-website/src/pages/index.md b/docs/my-website/src/pages/index.md
index 1dc2995c5fe..91215b33c5d 100644
--- a/docs/my-website/src/pages/index.md
+++ b/docs/my-website/src/pages/index.md
@@ -604,7 +604,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ docker.litellm.ai/berriai/litellm:main-latest \
--config /app/config.yaml --detailed_debug
```
diff --git a/litellm-proxy-extras/litellm_proxy_extras/utils.py b/litellm-proxy-extras/litellm_proxy_extras/utils.py
index 96e1a5106ac..7ffbe95be13 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/utils.py
+++ b/litellm-proxy-extras/litellm_proxy_extras/utils.py
@@ -18,6 +18,45 @@ def str_to_bool(value: Optional[str]) -> bool:
return value.lower() in ("true", "1", "t", "y", "yes")
+
+def _get_prisma_env() -> dict:
+ """Get environment variables for Prisma, handling offline mode if configured."""
+ prisma_env = os.environ.copy()
+ if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
+ # These env vars prevent Prisma from attempting downloads
+ prisma_env["NPM_CONFIG_PREFER_OFFLINE"] = "true"
+ prisma_env["NPM_CONFIG_CACHE"] = os.getenv("NPM_CONFIG_CACHE", "/app/.cache/npm")
+ return prisma_env
+
+
+def _get_prisma_command() -> str:
+ """Get the Prisma command to use, bypassing Python wrapper in offline mode."""
+ if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
+ # Primary location where Prisma Python package installs the CLI
+ default_cli_path = "/app/.cache/prisma-python/binaries/node_modules/.bin/prisma"
+
+ # Check if custom path is provided (for flexibility)
+ custom_cli_path = os.getenv("PRISMA_CLI_PATH")
+ if custom_cli_path and os.path.exists(custom_cli_path):
+ logger.info(f"Using custom Prisma CLI at {custom_cli_path}")
+ return custom_cli_path
+
+ # Check the default location
+ if os.path.exists(default_cli_path):
+ logger.info(f"Using cached Prisma CLI at {default_cli_path}")
+ return default_cli_path
+
+ # If not found, log warning and fall back
+ logger.warning(
+ f"Prisma CLI not found at {default_cli_path}. "
+ "Falling back to Python wrapper (may attempt downloads)"
+ )
+
+ # Fall back to the Python wrapper (will work in online mode)
+ return "prisma"
+
+
+
class ProxyExtrasDBManager:
@staticmethod
def _get_prisma_dir() -> str:
@@ -57,6 +96,11 @@ class ProxyExtrasDBManager:
init_dir.mkdir(parents=True, exist_ok=True)
database_url = os.getenv("DATABASE_URL")
+ if not database_url:
+ logger.error("DATABASE_URL not set")
+ return False
+ # Set up environment for offline mode if configured
+ prisma_env = _get_prisma_env()
try:
# 1. Generate migration SQL file by comparing empty state to current db state
@@ -64,7 +108,7 @@ class ProxyExtrasDBManager:
migration_file = init_dir / "migration.sql"
subprocess.run(
[
- "prisma",
+ _get_prisma_command(),
"migrate",
"diff",
"--from-empty",
@@ -75,13 +119,14 @@ class ProxyExtrasDBManager:
stdout=open(migration_file, "w"),
check=True,
timeout=30,
+ env=prisma_env
)
# 3. Mark the migration as applied since it represents current state
logger.info("Marking baseline migration as applied...")
subprocess.run(
[
- "prisma",
+ _get_prisma_command(),
"migrate",
"resolve",
"--applied",
@@ -89,6 +134,7 @@ class ProxyExtrasDBManager:
],
check=True,
timeout=30,
+ env=prisma_env
)
return True
@@ -113,21 +159,26 @@ class ProxyExtrasDBManager:
@staticmethod
def _roll_back_migration(migration_name: str):
"""Mark a specific migration as rolled back"""
+ # Set up environment for offline mode if configured
+ prisma_env = _get_prisma_env()
subprocess.run(
- ["prisma", "migrate", "resolve", "--rolled-back", migration_name],
+ [_get_prisma_command(), "migrate", "resolve", "--rolled-back", migration_name],
timeout=60,
check=True,
capture_output=True,
+ env=prisma_env
)
@staticmethod
def _resolve_specific_migration(migration_name: str):
"""Mark a specific migration as applied"""
+ prisma_env = _get_prisma_env()
subprocess.run(
- ["prisma", "migrate", "resolve", "--applied", migration_name],
+ [_get_prisma_command(), "migrate", "resolve", "--applied", migration_name],
timeout=60,
check=True,
capture_output=True,
+ env=prisma_env
)
@staticmethod
@@ -194,6 +245,10 @@ class ProxyExtrasDBManager:
3. Mark all existing migrations as applied.
"""
database_url = os.getenv("DATABASE_URL")
+ if not database_url:
+ logger.error("DATABASE_URL not set")
+ return
+
diff_dir = (
Path(migrations_dir)
/ "migrations"
@@ -216,7 +271,7 @@ class ProxyExtrasDBManager:
with open(diff_sql_path, "w") as f:
subprocess.run(
[
- "prisma",
+ _get_prisma_command(),
"migrate",
"diff",
"--from-url",
@@ -228,6 +283,7 @@ class ProxyExtrasDBManager:
check=True,
timeout=60,
stdout=f,
+ env=_get_prisma_env()
)
except subprocess.CalledProcessError as e:
logger.warning(f"Failed to generate migration diff: {e.stderr}")
@@ -245,7 +301,7 @@ class ProxyExtrasDBManager:
logger.info("Running prisma db execute to apply the migration diff...")
result = subprocess.run(
[
- "prisma",
+ _get_prisma_command(),
"db",
"execute",
"--file",
@@ -257,6 +313,7 @@ class ProxyExtrasDBManager:
check=True,
capture_output=True,
text=True,
+ env=_get_prisma_env()
)
logger.info(f"prisma db execute stdout: {result.stdout}")
logger.info("✅ Migration diff applied successfully")
@@ -274,11 +331,12 @@ class ProxyExtrasDBManager:
try:
logger.info(f"Resolving migration: {migration_name}")
subprocess.run(
- ["prisma", "migrate", "resolve", "--applied", migration_name],
+ [_get_prisma_command(), "migrate", "resolve", "--applied", migration_name],
timeout=60,
check=True,
capture_output=True,
text=True,
+ env=_get_prisma_env()
)
logger.debug(f"Resolved migration: {migration_name}")
except subprocess.CalledProcessError as e:
@@ -312,11 +370,12 @@ class ProxyExtrasDBManager:
try:
# Set migrations directory for Prisma
result = subprocess.run(
- ["prisma", "migrate", "deploy"],
+ [_get_prisma_command(), "migrate", "deploy"],
timeout=60,
check=True,
capture_output=True,
text=True,
+ env=_get_prisma_env()
)
logger.info(f"prisma migrate deploy stdout: {result.stdout}")
@@ -344,7 +403,7 @@ class ProxyExtrasDBManager:
# Mark the failed migration as rolled back
subprocess.run(
[
- "prisma",
+ _get_prisma_command(),
"migrate",
"resolve",
"--rolled-back",
@@ -354,6 +413,7 @@ class ProxyExtrasDBManager:
check=True,
capture_output=True,
text=True,
+ env=_get_prisma_env()
)
logger.info(
f"✅ Migration {failed_migration} marked as rolled back... retrying"
@@ -450,7 +510,7 @@ class ProxyExtrasDBManager:
else:
# Use prisma db push with increased timeout
subprocess.run(
- ["prisma", "db", "push", "--accept-data-loss"],
+ [_get_prisma_command(), "db", "push", "--accept-data-loss"],
timeout=60,
check=True,
)
diff --git a/litellm/__init__.py b/litellm/__init__.py
index 80625e0b189..42ae3c638ac 100644
--- a/litellm/__init__.py
+++ b/litellm/__init__.py
@@ -26,16 +26,6 @@ from typing import (
)
from litellm.types.integrations.datadog_llm_obs import DatadogLLMObsInitParams
from litellm.types.integrations.datadog import DatadogInitParams
-from litellm.caching.llm_caching_handler import LLMClientCache
-from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES
-from litellm.types.utils import (
- ImageObject,
- BudgetConfig,
- all_litellm_params,
- all_litellm_params as _litellm_completion_params,
- CredentialItem,
- PriorityReservationDict,
-) # maintain backwards compatibility for root param.
from litellm._logging import (
set_verbose,
_turn_on_debug,
@@ -82,11 +72,6 @@ from litellm.constants import (
DEFAULT_SOFT_BUDGET,
DEFAULT_ALLOWED_FAILS,
)
-from litellm.integrations.dotprompt import (
- global_prompt_manager,
- global_prompt_directory,
- set_global_prompt_directory,
-)
from litellm.types.guardrails import GuardrailItem
from litellm.types.secret_managers.main import (
KeyManagementSystem,
@@ -96,11 +81,7 @@ from litellm.types.proxy.management_endpoints.ui_sso import (
DefaultTeamSSOParams,
LiteLLM_UpperboundKeyGenerateParams,
)
-from litellm.types.utils import (
- StandardKeyGenerationConfig,
- LlmProviders,
- SearchProviders,
-)
+from litellm.types.utils import LlmProviders
from litellm.types.utils import PriorityReservationSettings
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.logging_callback_manager import LoggingCallbackManager
@@ -285,7 +266,7 @@ disable_token_counter: bool = False
disable_add_transform_inline_image_block: bool = False
disable_add_user_agent_to_request_tags: bool = False
extra_spend_tag_headers: Optional[List[str]] = None
-in_memory_llm_clients_cache: LLMClientCache = LLMClientCache()
+in_memory_llm_clients_cache: "LLMClientCache"
safe_memory_mode: bool = False
enable_azure_ad_token_refresh: Optional[bool] = False
### DEFAULT AZURE API VERSION ###
@@ -293,9 +274,9 @@ AZURE_DEFAULT_API_VERSION = "2025-02-01-preview" # this is updated to the lates
### DEFAULT WATSONX API VERSION ###
WATSONX_DEFAULT_API_VERSION = "2024-03-13"
### COHERE EMBEDDINGS DEFAULT TYPE ###
-COHERE_DEFAULT_EMBEDDING_INPUT_TYPE: COHERE_EMBEDDING_INPUT_TYPES = "search_document"
+COHERE_DEFAULT_EMBEDDING_INPUT_TYPE: "COHERE_EMBEDDING_INPUT_TYPES" = "search_document"
### CREDENTIALS ###
-credential_list: List[CredentialItem] = []
+credential_list: List["CredentialItem"] = []
### GUARDRAILS ###
llamaguard_model_name: Optional[str] = None
openai_moderations_model_name: Optional[str] = None
@@ -370,7 +351,7 @@ aws_sqs_callback_params: Optional[Dict] = None
generic_logger_headers: Optional[Dict] = None
default_key_generate_params: Optional[Dict] = None
upperbound_key_generate_params: Optional[LiteLLM_UpperboundKeyGenerateParams] = None
-key_generation_settings: Optional[StandardKeyGenerationConfig] = None
+key_generation_settings: Optional["StandardKeyGenerationConfig"] = None
default_internal_user_params: Optional[Dict] = None
default_team_params: Optional[Union[DefaultTeamSSOParams, Dict]] = None
default_team_settings: Optional[List] = None
@@ -379,7 +360,7 @@ default_max_internal_user_budget: Optional[float] = None
max_internal_user_budget: Optional[float] = None
max_ui_session_budget: Optional[float] = 10 # $10 USD budgets for UI Chat sessions
internal_user_budget_duration: Optional[str] = None
-tag_budget_config: Optional[Dict[str, BudgetConfig]] = None
+tag_budget_config: Optional[Dict[str, "BudgetConfig"]] = None
max_end_user_budget: Optional[float] = None
max_end_user_budget_id: Optional[str] = None
disable_end_user_cost_tracking: Optional[bool] = None
@@ -402,7 +383,9 @@ public_agent_groups: Optional[List[str]] = None
# Old format: { "displayName": "url" } (for backward compatibility)
public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {}
#### REQUEST PRIORITIZATION #######
-priority_reservation: Optional[Dict[str, Union[float, PriorityReservationDict]]] = None
+priority_reservation: Optional[
+ Dict[str, Union[float, "PriorityReservationDict"]]
+] = None
priority_reservation_settings: "PriorityReservationSettings" = (
PriorityReservationSettings()
)
@@ -1471,7 +1454,6 @@ from . import rag
### CUSTOM LLMs ###
from .types.llms.custom_llm import CustomLLMItem
-from .types.utils import GenericStreamingChunk
custom_provider_map: List[CustomLLMItem] = []
_custom_providers: List[str] = (
@@ -1515,6 +1497,14 @@ if TYPE_CHECKING:
from litellm.types.utils import ModelInfo as _ModelInfoType
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.caching.caching import Cache
+ from litellm.caching.llm_caching_handler import LLMClientCache
+ from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES
+ from litellm.types.utils import (
+ BudgetConfig,
+ CredentialItem,
+ PriorityReservationDict,
+ StandardKeyGenerationConfig,
+ )
# Cost calculator functions
cost_per_token: Callable[..., Tuple[float, float]]
@@ -1567,8 +1557,12 @@ def __getattr__(name: str) -> Any:
LITELLM_LOGGING_NAMES,
UTILS_NAMES,
TOKEN_COUNTER_NAMES,
+ LLM_CLIENT_CACHE_NAMES,
+ BEDROCK_TYPES_NAMES,
+ TYPES_UTILS_NAMES,
CACHING_NAMES,
HTTP_HANDLER_NAMES,
+ DOTPROMPT_NAMES,
)
# Lazy load cost_calculator functions
@@ -1591,6 +1585,21 @@ def __getattr__(name: str) -> Any:
from ._lazy_imports import _lazy_import_token_counter
return _lazy_import_token_counter(name)
+ # Lazy load Bedrock type aliases
+ if name in BEDROCK_TYPES_NAMES:
+ from ._lazy_imports import _lazy_import_bedrock_types
+ return _lazy_import_bedrock_types(name)
+
+ # Lazy load common types.utils symbols
+ if name in TYPES_UTILS_NAMES:
+ from ._lazy_imports import _lazy_import_types_utils
+ return _lazy_import_types_utils(name)
+
+ # Lazy load LLM client cache and its singleton
+ if name in LLM_CLIENT_CACHE_NAMES:
+ from ._lazy_imports import _lazy_import_llm_client_cache
+ return _lazy_import_llm_client_cache(name)
+
# Lazy load caching classes
if name in CACHING_NAMES:
from ._lazy_imports import _lazy_import_caching
@@ -1602,6 +1611,12 @@ def __getattr__(name: str) -> Any:
return _lazy_import_http_handlers(name)
+ # Lazy load dotprompt integration globals
+ if name in DOTPROMPT_NAMES:
+ from ._lazy_imports import _lazy_import_dotprompt
+
+ return _lazy_import_dotprompt(name)
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
diff --git a/litellm/_lazy_imports.py b/litellm/_lazy_imports.py
index 1fbf3f1be0f..2c495aa8db5 100644
--- a/litellm/_lazy_imports.py
+++ b/litellm/_lazy_imports.py
@@ -1,10 +1,31 @@
-from typing import Any, cast
+from typing import Any, Optional, cast
import sys
def _get_litellm_globals() -> dict:
"""Helper to get the globals dictionary of the litellm module."""
return sys.modules["litellm"].__dict__
+# Lazy loader for default encoding to avoid importing tiktoken at module import time
+_default_encoding: Optional[Any] = None
+
+
+def _get_default_encoding() -> Any:
+ """
+ Lazily load and cache the default OpenAI encoding.
+
+ This avoids importing `litellm.litellm_core_utils.default_encoding` (and thus tiktoken)
+ at `litellm` import time. The encoding is cached after the first import.
+
+ This is used internally by utils.py functions that need the encoding but shouldn't
+ trigger its import during module load.
+ """
+ global _default_encoding
+ if _default_encoding is None:
+ from litellm.litellm_core_utils.default_encoding import encoding
+
+ _default_encoding = encoding
+ return _default_encoding
+
# Cost calculator names that support lazy loading via _lazy_import_cost_calculator
COST_CALCULATOR_NAMES = (
"completion_cost",
@@ -39,6 +60,31 @@ TOKEN_COUNTER_NAMES = (
"get_modified_max_tokens",
)
+# LLM client cache names that support lazy loading via _lazy_import_llm_client_cache
+LLM_CLIENT_CACHE_NAMES = (
+ "LLMClientCache",
+ "in_memory_llm_clients_cache",
+)
+
+# Bedrock type names that support lazy loading via _lazy_import_bedrock_types
+BEDROCK_TYPES_NAMES = (
+ "COHERE_EMBEDDING_INPUT_TYPES",
+)
+
+# Common types from litellm.types.utils that support lazy loading via
+# _lazy_import_types_utils
+TYPES_UTILS_NAMES = (
+ "ImageObject",
+ "BudgetConfig",
+ "all_litellm_params",
+ "_litellm_completion_params",
+ "CredentialItem",
+ "PriorityReservationDict",
+ "StandardKeyGenerationConfig",
+ "SearchProviders",
+ "GenericStreamingChunk",
+)
+
# Caching / cache classes that support lazy loading via _lazy_import_caching
CACHING_NAMES = (
"Cache",
@@ -53,6 +99,13 @@ HTTP_HANDLER_NAMES = (
"module_level_client",
)
+# Dotprompt integration names that support lazy loading via _lazy_import_dotprompt
+DOTPROMPT_NAMES = (
+ "global_prompt_manager",
+ "global_prompt_directory",
+ "set_global_prompt_directory",
+)
+
# Lazy import for utils module - imports only the requested item by name.
# Note: PLR0915 (too many statements) is suppressed because the many if statements
# are intentional - each attribute is imported individually only when requested,
@@ -299,6 +352,88 @@ def _lazy_import_token_counter(name: str) -> Any:
raise AttributeError(f"Token counter lazy import: unknown attribute {name!r}")
+def _lazy_import_bedrock_types(name: str) -> Any:
+ """Lazy import for Bedrock type aliases."""
+ _globals = _get_litellm_globals()
+
+ if name == "COHERE_EMBEDDING_INPUT_TYPES":
+ from litellm.types.llms.bedrock import (
+ COHERE_EMBEDDING_INPUT_TYPES as _COHERE_EMBEDDING_INPUT_TYPES,
+ )
+
+ _globals["COHERE_EMBEDDING_INPUT_TYPES"] = _COHERE_EMBEDDING_INPUT_TYPES
+ return _COHERE_EMBEDDING_INPUT_TYPES
+
+ raise AttributeError(f"Bedrock types lazy import: unknown attribute {name!r}")
+
+
+def _lazy_import_types_utils(name: str) -> Any:
+ """Lazy import for common types and constants from litellm.types.utils."""
+ _globals = _get_litellm_globals()
+
+ if name == "ImageObject":
+ from .types.utils import ImageObject as _ImageObject
+
+ _globals["ImageObject"] = _ImageObject
+ return _ImageObject
+
+ if name == "BudgetConfig":
+ from .types.utils import BudgetConfig as _BudgetConfig
+
+ _globals["BudgetConfig"] = _BudgetConfig
+ return _BudgetConfig
+
+ if name == "all_litellm_params":
+ from .types.utils import all_litellm_params as _all_litellm_params
+
+ _globals["all_litellm_params"] = _all_litellm_params
+ return _all_litellm_params
+
+ if name == "_litellm_completion_params":
+ from .types.utils import all_litellm_params as _all_litellm_params
+
+ _globals["_litellm_completion_params"] = _all_litellm_params
+ return _all_litellm_params
+
+ if name == "CredentialItem":
+ from .types.utils import CredentialItem as _CredentialItem
+
+ _globals["CredentialItem"] = _CredentialItem
+ return _CredentialItem
+
+ if name == "PriorityReservationDict":
+ from .types.utils import (
+ PriorityReservationDict as _PriorityReservationDict,
+ )
+
+ _globals["PriorityReservationDict"] = _PriorityReservationDict
+ return _PriorityReservationDict
+
+ if name == "StandardKeyGenerationConfig":
+ from .types.utils import (
+ StandardKeyGenerationConfig as _StandardKeyGenerationConfig,
+ )
+
+ _globals["StandardKeyGenerationConfig"] = _StandardKeyGenerationConfig
+ return _StandardKeyGenerationConfig
+
+ if name == "SearchProviders":
+ from .types.utils import SearchProviders as _SearchProviders
+
+ _globals["SearchProviders"] = _SearchProviders
+ return _SearchProviders
+
+ if name == "GenericStreamingChunk":
+ from .types.utils import (
+ GenericStreamingChunk as _GenericStreamingChunk,
+ )
+
+ _globals["GenericStreamingChunk"] = _GenericStreamingChunk
+ return _GenericStreamingChunk
+
+ raise AttributeError(f"Types utils lazy import: unknown attribute {name!r}")
+
+
def _lazy_import_caching(name: str) -> Any:
"""Lazy import for caching module classes."""
_globals = _get_litellm_globals()
@@ -330,6 +465,28 @@ def _lazy_import_caching(name: str) -> Any:
raise AttributeError(f"Caching lazy import: unknown attribute {name!r}")
+def _lazy_import_llm_client_cache(name: str) -> Any:
+ """Lazy import for LLM client cache class and singleton."""
+ _globals = _get_litellm_globals()
+
+ if name == "LLMClientCache":
+ from litellm.caching.llm_caching_handler import LLMClientCache as _LLMClientCache
+
+ _globals["LLMClientCache"] = _LLMClientCache
+ return _LLMClientCache
+
+ if name == "in_memory_llm_clients_cache":
+ from litellm.caching.llm_caching_handler import LLMClientCache as _LLMClientCache
+
+ instance = _LLMClientCache()
+ # Only populate the requested singleton name to keep lazy-import
+ # semantics consistent with other helpers (no extra symbols).
+ _globals["in_memory_llm_clients_cache"] = instance
+ return instance
+
+ raise AttributeError(f"LLM client cache lazy import: unknown attribute {name!r}")
+
+
def _lazy_import_litellm_logging(name: str) -> Any:
"""Lazy import for litellm_logging module."""
_globals = _get_litellm_globals()
@@ -375,4 +532,35 @@ def _lazy_import_http_handlers(name: str) -> Any:
_globals["module_level_client"] = sync_client
return sync_client
- raise AttributeError(f"HTTP handlers lazy import: unknown attribute {name!r}")
\ No newline at end of file
+ raise AttributeError(f"HTTP handlers lazy import: unknown attribute {name!r}")
+
+
+def _lazy_import_dotprompt(name: str) -> Any:
+ """Lazy import for dotprompt integration globals."""
+ _globals = _get_litellm_globals()
+
+ if name == "global_prompt_manager":
+ from litellm.integrations.dotprompt import (
+ global_prompt_manager as _global_prompt_manager,
+ )
+
+ _globals["global_prompt_manager"] = _global_prompt_manager
+ return _global_prompt_manager
+
+ if name == "global_prompt_directory":
+ from litellm.integrations.dotprompt import (
+ global_prompt_directory as _global_prompt_directory,
+ )
+
+ _globals["global_prompt_directory"] = _global_prompt_directory
+ return _global_prompt_directory
+
+ if name == "set_global_prompt_directory":
+ from litellm.integrations.dotprompt import (
+ set_global_prompt_directory as _set_global_prompt_directory,
+ )
+
+ _globals["set_global_prompt_directory"] = _set_global_prompt_directory
+ return _set_global_prompt_directory
+
+ raise AttributeError(f"Dotprompt lazy import: unknown attribute {name!r}")
\ No newline at end of file
diff --git a/litellm/a2a_protocol/providers/pydantic_ai_agents/transformation.py b/litellm/a2a_protocol/providers/pydantic_ai_agents/transformation.py
index 6f46933cf9f..9352eab6c8e 100644
--- a/litellm/a2a_protocol/providers/pydantic_ai_agents/transformation.py
+++ b/litellm/a2a_protocol/providers/pydantic_ai_agents/transformation.py
@@ -6,12 +6,11 @@ This module provides fake streaming by converting non-streaming responses into s
"""
import asyncio
-from typing import Any, AsyncIterator, Dict
+from typing import Any, AsyncIterator, Dict, cast
from uuid import uuid4
-import httpx
-
from litellm._logging import verbose_logger
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, get_async_httpx_client
class PydanticAITransformation:
@@ -78,7 +77,7 @@ class PydanticAITransformation:
@staticmethod
async def _poll_for_completion(
- client: httpx.AsyncClient,
+ client: AsyncHTTPHandler,
endpoint: str,
task_id: str,
request_id: str,
@@ -179,34 +178,37 @@ class PydanticAITransformation:
f"Pydantic AI: Sending non-streaming request to {endpoint}"
)
- # Send request to Pydantic AI agent
- async with httpx.AsyncClient(timeout=timeout) as client:
- response = await client.post(
- endpoint,
- json=a2a_request,
- headers={"Content-Type": "application/json"},
- )
- response.raise_for_status()
- response_data = response.json()
-
- # Check if task is already completed
- result = response_data.get("result", {})
- status = result.get("status", {})
- state = status.get("state", "")
-
- if state != "completed":
- # Need to poll for completion
- task_id = result.get("id")
- if task_id:
- verbose_logger.info(
- f"Pydantic AI: Task {task_id} submitted, polling for completion..."
- )
- response_data = await PydanticAITransformation._poll_for_completion(
- client=client,
- endpoint=endpoint,
- task_id=task_id,
- request_id=request_id,
- )
+ # Send request to Pydantic AI agent using shared async HTTP client
+ client = get_async_httpx_client(
+ llm_provider=cast(Any, "pydantic_ai_agent"),
+ params={"timeout": timeout},
+ )
+ response = await client.post(
+ endpoint,
+ json=a2a_request,
+ headers={"Content-Type": "application/json"},
+ )
+ response.raise_for_status()
+ response_data = response.json()
+
+ # Check if task is already completed
+ result = response_data.get("result", {})
+ status = result.get("status", {})
+ state = status.get("state", "")
+
+ if state != "completed":
+ # Need to poll for completion
+ task_id = result.get("id")
+ if task_id:
+ verbose_logger.info(
+ f"Pydantic AI: Task {task_id} submitted, polling for completion..."
+ )
+ response_data = await PydanticAITransformation._poll_for_completion(
+ client=client,
+ endpoint=endpoint,
+ task_id=task_id,
+ request_id=request_id,
+ )
verbose_logger.info(f"Pydantic AI: Received completed response for request_id={request_id}")
diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py
index 56035dad68d..612bec239ba 100644
--- a/litellm/completion_extras/litellm_responses_transformation/transformation.py
+++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py
@@ -457,6 +457,24 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
raw_response.usage
),
)
+
+ # Preserve hidden params from the ResponsesAPIResponse, especially the headers
+ # which contain important provider information like x-request-id
+ raw_response_hidden_params = getattr(raw_response, "_hidden_params", {})
+ if raw_response_hidden_params:
+ if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None:
+ model_response._hidden_params = {}
+ # Merge the raw_response hidden params with model_response hidden params
+ # Preserve existing keys in model_response but add/override with raw_response params
+ for key, value in raw_response_hidden_params.items():
+ if key == "additional_headers" and key in model_response._hidden_params:
+ # Merge additional_headers to preserve both sets
+ existing_additional_headers = model_response._hidden_params.get("additional_headers", {})
+ merged_headers = {**value, **existing_additional_headers}
+ model_response._hidden_params[key] = merged_headers
+ else:
+ model_response._hidden_params[key] = value
+
return model_response
def get_model_response_iterator(
diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py
index 6488128b215..6771999cd35 100644
--- a/litellm/integrations/custom_logger.py
+++ b/litellm/integrations/custom_logger.py
@@ -16,7 +16,6 @@ from typing import (
from pydantic import BaseModel
from litellm._logging import verbose_logger
-from litellm.caching.caching import DualCache
from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER
from litellm.types.integrations.argilla import ArgillaItem
from litellm.types.llms.openai import AllMessageValues, ChatCompletionRequest
@@ -33,6 +32,7 @@ from litellm.types.utils import (
)
if TYPE_CHECKING:
+ from litellm.caching.caching import DualCache
from opentelemetry.trace import Span as _Span
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
@@ -334,7 +334,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
async def async_pre_call_hook(
self,
user_api_key_dict: UserAPIKeyAuth,
- cache: DualCache,
+ cache: "DualCache",
data: dict,
call_type: CallTypesLiteral,
) -> Optional[
diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py
index 5a50806218f..59d2a8a8dd0 100644
--- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py
+++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py
@@ -430,6 +430,18 @@ def convert_to_model_response_object( # noqa: PLR0915
if hidden_params is None:
hidden_params = {}
+
+ # Preserve existing additional_headers if they contain important provider headers
+ # For responses API, additional_headers may already be set with LLM provider headers
+ existing_additional_headers = hidden_params.get("additional_headers", {})
+ if existing_additional_headers and _response_headers is None:
+ # Keep existing headers when _response_headers is None (responses API case)
+ additional_headers = existing_additional_headers
+ else:
+ # Merge new headers with existing ones
+ if existing_additional_headers:
+ additional_headers.update(existing_additional_headers)
+
hidden_params["additional_headers"] = additional_headers
### CHECK IF ERROR IN RESPONSE ### - openrouter returns these in the dictionary
diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py
index 90c8c30eed6..7649c276b32 100644
--- a/litellm/llms/anthropic/chat/handler.py
+++ b/litellm/llms/anthropic/chat/handler.py
@@ -340,7 +340,7 @@ class AnthropicChatCompletion(BaseLLM):
data = config.transform_request(
model=model,
messages=messages,
- optional_params=optional_params,
+ optional_params={**optional_params, "is_vertex_request": is_vertex_request},
litellm_params=litellm_params,
headers=headers,
)
diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py
index 628121ab11c..b88b7e71263 100644
--- a/litellm/llms/anthropic/chat/transformation.py
+++ b/litellm/llms/anthropic/chat/transformation.py
@@ -942,6 +942,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
self, headers: dict, optional_params: dict
) -> dict:
"""Update headers with optional anthropic beta."""
+
+ # Skip adding beta headers for Vertex requests
+ # Vertex AI handles these headers differently
+ is_vertex_request = optional_params.get("is_vertex_request", False)
+ if is_vertex_request:
+ return headers
_tools = optional_params.get("tools", [])
for tool in _tools:
@@ -1067,6 +1073,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
):
optional_params["metadata"] = {"user_id": _litellm_metadata["user_id"]}
+ # Remove internal LiteLLM parameters that should not be sent to Anthropic API
+ optional_params.pop("is_vertex_request", None)
+
data = {
"model": model,
"messages": anthropic_messages,
diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py
index 32be1a780a3..ca646254d5d 100644
--- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py
+++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py
@@ -108,6 +108,27 @@ class AmazonAnthropicClaudeMessagesConfig(
stream=stream,
)
+ def _remove_ttl_from_cache_control(
+ self, anthropic_messages_request: Dict
+ ) -> None:
+ """
+ Remove `ttl` field from cache_control in messages.
+ Bedrock doesn't support the ttl field in cache_control.
+
+ Args:
+ anthropic_messages_request: The request dictionary to modify in-place
+ """
+ if "messages" in anthropic_messages_request:
+ for message in anthropic_messages_request["messages"]:
+ if isinstance(message, dict) and "content" in message:
+ content = message["content"]
+ if isinstance(content, list):
+ for item in content:
+ if isinstance(item, dict) and "cache_control" in item:
+ cache_control = item["cache_control"]
+ if isinstance(cache_control, dict) and "ttl" in cache_control:
+ cache_control.pop("ttl", None)
+
def transform_anthropic_messages_request(
self,
model: str,
@@ -141,8 +162,11 @@ class AmazonAnthropicClaudeMessagesConfig(
# 3. `model` is not allowed in request body for bedrock invoke
if "model" in anthropic_messages_request:
anthropic_messages_request.pop("model", None)
+
+ # 4. Remove `ttl` field from cache_control in messages (Bedrock doesn't support it)
+ self._remove_ttl_from_cache_control(anthropic_messages_request)
- # 4. AUTO-INJECT beta headers based on features used
+ # 5. AUTO-INJECT beta headers based on features used
anthropic_model_info = AnthropicModelInfo()
tools = anthropic_messages_optional_request_params.get("tools")
messages_typed = cast(List[AllMessageValues], messages)
diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py
index 5697700b46d..1b4d04af80a 100644
--- a/litellm/llms/custom_httpx/http_handler.py
+++ b/litellm/llms/custom_httpx/http_handler.py
@@ -1153,7 +1153,17 @@ def get_async_httpx_client(
pass
_cache_key_name = "async_httpx_client" + _params_key_name + llm_provider
- _cached_client = litellm.in_memory_llm_clients_cache.get_cache(_cache_key_name)
+
+ # Lazily initialize the global in-memory client cache to avoid relying on
+ # litellm globals being fully populated during import time.
+ cache = getattr(litellm, "in_memory_llm_clients_cache", None)
+ if cache is None:
+ from litellm.caching.llm_caching_handler import LLMClientCache
+
+ cache = LLMClientCache()
+ setattr(litellm, "in_memory_llm_clients_cache", cache)
+
+ _cached_client = cache.get_cache(_cache_key_name)
if _cached_client:
return _cached_client
@@ -1166,7 +1176,7 @@ def get_async_httpx_client(
shared_session=shared_session,
)
- litellm.in_memory_llm_clients_cache.set_cache(
+ cache.set_cache(
key=_cache_key_name,
value=_new_client,
ttl=_DEFAULT_TTL_FOR_HTTPX_CLIENTS,
@@ -1191,7 +1201,16 @@ def _get_httpx_client(params: Optional[dict] = None) -> HTTPHandler:
_cache_key_name = "httpx_client" + _params_key_name
- _cached_client = litellm.in_memory_llm_clients_cache.get_cache(_cache_key_name)
+ # Lazily initialize the global in-memory client cache to avoid relying on
+ # litellm globals being fully populated during import time.
+ cache = getattr(litellm, "in_memory_llm_clients_cache", None)
+ if cache is None:
+ from litellm.caching.llm_caching_handler import LLMClientCache
+
+ cache = LLMClientCache()
+ setattr(litellm, "in_memory_llm_clients_cache", cache)
+
+ _cached_client = cache.get_cache(_cache_key_name)
if _cached_client:
return _cached_client
@@ -1200,7 +1219,7 @@ def _get_httpx_client(params: Optional[dict] = None) -> HTTPHandler:
else:
_new_client = HTTPHandler(timeout=httpx.Timeout(timeout=600.0, connect=5.0))
- litellm.in_memory_llm_clients_cache.set_cache(
+ cache.set_cache(
key=_cache_key_name,
value=_new_client,
ttl=_DEFAULT_TTL_FOR_HTTPX_CLIENTS,
diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py
index 4c9d3828383..7ccec074703 100644
--- a/litellm/llms/openai/responses/transformation.py
+++ b/litellm/llms/openai/responses/transformation.py
@@ -6,6 +6,7 @@ from pydantic import BaseModel
import litellm
from litellm._logging import verbose_logger
+from litellm.litellm_core_utils.core_helpers import process_response_headers
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
_safe_convert_created_field,
)
@@ -15,7 +16,7 @@ from litellm.types.llms.openai import *
from litellm.types.responses.main import *
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
-from litellm.litellm_core_utils.core_helpers import process_response_headers
+
from ..common_utils import OpenAIError
if TYPE_CHECKING:
@@ -181,6 +182,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
)
response = ResponsesAPIResponse.model_construct(**raw_response_json)
+ # Store processed headers in additional_headers so they get returned to the client
response._hidden_params["additional_headers"] = processed_headers
response._hidden_params["headers"] = raw_response_headers
return response
diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json
index a6c19222619..2d801506d5f 100644
--- a/litellm/llms/openai_like/providers.json
+++ b/litellm/llms/openai_like/providers.json
@@ -14,5 +14,9 @@
"helicone": {
"base_url": "https://ai-gateway.helicone.ai/",
"api_key_env": "HELICONE_API_KEY"
+ },
+ "veniceai": {
+ "base_url": "https://api.venice.ai/api/v1",
+ "api_key_env": "VENICE_AI_API_KEY"
}
}
diff --git a/litellm/llms/vertex_ai/agent_engine/__init__.py b/litellm/llms/vertex_ai/agent_engine/__init__.py
new file mode 100644
index 00000000000..de891f85602
--- /dev/null
+++ b/litellm/llms/vertex_ai/agent_engine/__init__.py
@@ -0,0 +1,13 @@
+"""
+Vertex AI Agent Engine (Reasoning Engines) Provider
+
+Supports Vertex AI Reasoning Engines via the :query and :streamQuery endpoints.
+"""
+
+from litellm.llms.vertex_ai.agent_engine.transformation import (
+ VertexAgentEngineConfig,
+ VertexAgentEngineError,
+)
+
+__all__ = ["VertexAgentEngineConfig", "VertexAgentEngineError"]
+
diff --git a/litellm/llms/vertex_ai/agent_engine/sse_iterator.py b/litellm/llms/vertex_ai/agent_engine/sse_iterator.py
new file mode 100644
index 00000000000..06fb55e1848
--- /dev/null
+++ b/litellm/llms/vertex_ai/agent_engine/sse_iterator.py
@@ -0,0 +1,90 @@
+"""
+SSE Stream Iterator for Vertex AI Agent Engine.
+
+Handles Server-Sent Events (SSE) streaming responses from Vertex AI Reasoning Engines.
+"""
+
+from typing import Any, Union
+
+from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
+from litellm.types.llms.openai import ChatCompletionUsageBlock
+from litellm.types.utils import (
+ Delta,
+ GenericStreamingChunk,
+ ModelResponseStream,
+ StreamingChoices,
+)
+
+
+class VertexAgentEngineResponseIterator(BaseModelResponseIterator):
+ """
+ Iterator for Vertex Agent Engine SSE streaming responses.
+
+ Uses BaseModelResponseIterator which handles sync/async iteration.
+ We just need to implement chunk_parser to parse Vertex Agent Engine response format.
+ """
+
+ def __init__(self, streaming_response: Any, sync_stream: bool) -> None:
+ super().__init__(streaming_response=streaming_response, sync_stream=sync_stream)
+
+ def chunk_parser(
+ self, chunk: dict
+ ) -> Union[GenericStreamingChunk, ModelResponseStream]:
+ """
+ Parse a Vertex Agent Engine response chunk into ModelResponseStream.
+
+ Vertex Agent Engine response format:
+ {
+ "content": {
+ "parts": [{"text": "..."}],
+ "role": "model"
+ },
+ "finish_reason": "STOP",
+ "usage_metadata": {
+ "prompt_token_count": 100,
+ "candidates_token_count": 50,
+ "total_token_count": 150
+ }
+ }
+ """
+ # Extract text from content.parts
+ text = None
+ content = chunk.get("content", {})
+ parts = content.get("parts", [])
+ for part in parts:
+ if isinstance(part, dict) and "text" in part:
+ text = part["text"]
+ break
+
+ # Extract finish_reason
+ finish_reason = None
+ raw_finish_reason = chunk.get("finish_reason")
+ if raw_finish_reason == "STOP":
+ finish_reason = "stop"
+ elif raw_finish_reason:
+ finish_reason = raw_finish_reason.lower()
+
+ # Extract usage from usage_metadata
+ usage = None
+ usage_metadata = chunk.get("usage_metadata", {})
+ if usage_metadata:
+ usage = ChatCompletionUsageBlock(
+ prompt_tokens=usage_metadata.get("prompt_token_count", 0),
+ completion_tokens=usage_metadata.get("candidates_token_count", 0),
+ total_tokens=usage_metadata.get("total_token_count", 0),
+ )
+
+ # Return ModelResponseStream (OpenAI-compatible chunk)
+ return ModelResponseStream(
+ choices=[
+ StreamingChoices(
+ finish_reason=finish_reason,
+ index=0,
+ delta=Delta(
+ content=text,
+ role="assistant" if text else None,
+ ),
+ )
+ ],
+ usage=usage,
+ )
diff --git a/litellm/llms/vertex_ai/agent_engine/transformation.py b/litellm/llms/vertex_ai/agent_engine/transformation.py
new file mode 100644
index 00000000000..4c07e8455e3
--- /dev/null
+++ b/litellm/llms/vertex_ai/agent_engine/transformation.py
@@ -0,0 +1,508 @@
+"""
+Transformation for Vertex AI Agent Engine (Reasoning Engines)
+
+Handles the transformation between LiteLLM's OpenAI-compatible format and
+Vertex AI Reasoning Engine's API format.
+
+API Reference:
+- :query endpoint - for session management (create, get, list, delete)
+- :streamQuery endpoint - for actual queries (stream_query method)
+"""
+
+import json
+from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast
+
+import httpx
+
+from litellm._logging import verbose_logger
+from litellm._uuid import uuid
+from litellm.litellm_core_utils.prompt_templates.common_utils import (
+ convert_content_list_to_str,
+)
+from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
+from litellm.llms.vertex_ai.agent_engine.sse_iterator import (
+ VertexAgentEngineResponseIterator,
+)
+from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
+from litellm.types.llms.openai import AllMessageValues
+from litellm.types.utils import Choices, Message, ModelResponse, Usage
+
+if TYPE_CHECKING:
+ from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
+ from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
+ from litellm.utils import CustomStreamWrapper
+
+ LiteLLMLoggingObj = _LiteLLMLoggingObj
+else:
+ LiteLLMLoggingObj = Any
+ HTTPHandler = Any
+ AsyncHTTPHandler = Any
+ CustomStreamWrapper = Any
+
+
+class VertexAgentEngineError(BaseLLMException):
+ """Exception for Vertex Agent Engine errors."""
+
+ def __init__(self, status_code: int, message: str):
+ self.status_code = status_code
+ self.message = message
+ super().__init__(message=message, status_code=status_code)
+
+
+class VertexAgentEngineConfig(BaseConfig, VertexBase):
+ """
+ Configuration for Vertex AI Agent Engine (Reasoning Engines).
+
+ Model format: vertex_ai/agent_engine/
+ Where resource_id is the numeric ID of the reasoning engine.
+ """
+
+ def __init__(self, **kwargs):
+ BaseConfig.__init__(self, **kwargs)
+ VertexBase.__init__(self)
+
+ def get_supported_openai_params(self, model: str) -> List[str]:
+ """Vertex Agent Engine has limited OpenAI compatible params."""
+ return ["user"]
+
+ def map_openai_params(
+ self,
+ non_default_params: dict,
+ optional_params: dict,
+ model: str,
+ drop_params: bool,
+ ) -> dict:
+ """Map OpenAI params to Agent Engine params."""
+ # Map 'user' to 'user_id' for session management
+ if "user" in non_default_params:
+ optional_params["user_id"] = non_default_params["user"]
+ return optional_params
+
+ def _parse_model_string(self, model: str) -> Tuple[str, str]:
+ """
+ Parse model string to extract resource ID.
+
+ Model format: agent_engine///
+ Or: agent_engine/ (uses default project/location)
+
+ Returns: (resource_path, engine_id)
+ """
+ # Remove 'agent_engine/' prefix if present
+ if model.startswith("agent_engine/"):
+ model = model[len("agent_engine/") :]
+
+ # Check if it's a full resource path
+ if model.startswith("projects/"):
+ # Full path: projects/123/locations/us-central1/reasoningEngines/456
+ return model, model.split("/")[-1]
+
+ # Just the engine ID
+ return model, model
+
+ def get_complete_url(
+ self,
+ api_base: Optional[str],
+ api_key: Optional[str],
+ model: str,
+ optional_params: dict,
+ litellm_params: dict,
+ stream: Optional[bool] = None,
+ ) -> str:
+ """
+ Get the complete URL for the request.
+
+ For Vertex Agent Engine:
+ - Non-streaming: :query endpoint (for session management)
+ - Streaming: :streamQuery endpoint (for actual queries)
+ """
+ resource_path, engine_id = self._parse_model_string(model)
+
+ # Get project and location from litellm_params or environment
+ vertex_project = self.safe_get_vertex_ai_project(litellm_params)
+ vertex_location = self.safe_get_vertex_ai_location(litellm_params) or "us-central1"
+
+ # Build the full resource path if only engine_id was provided
+ if not resource_path.startswith("projects/"):
+ if not vertex_project:
+ raise ValueError(
+ "vertex_project is required for Vertex Agent Engine. "
+ "Set via litellm_params['vertex_project'] or VERTEXAI_PROJECT env var."
+ )
+ resource_path = f"projects/{vertex_project}/locations/{vertex_location}/reasoningEngines/{engine_id}"
+
+ # Build the base URL
+ base_url = f"https://{vertex_location}-aiplatform.googleapis.com"
+
+ # Always use :streamQuery endpoint for actual queries
+ # The :query endpoint only supports session management methods
+ # (create_session, get_session, list_sessions, delete_session, etc.)
+ endpoint = f"{base_url}/v1beta1/{resource_path}:streamQuery"
+
+ verbose_logger.debug(f"Vertex Agent Engine URL: {endpoint}")
+ return endpoint
+
+ def _get_auth_headers(
+ self,
+ optional_params: dict,
+ litellm_params: dict,
+ ) -> Dict[str, str]:
+ """Get authentication headers using Google Cloud credentials."""
+ vertex_credentials = self.safe_get_vertex_ai_credentials(litellm_params)
+ vertex_project = self.safe_get_vertex_ai_project(litellm_params)
+
+ # Get access token using VertexBase
+ access_token, project_id = self.get_access_token(
+ credentials=vertex_credentials,
+ project_id=vertex_project,
+ )
+
+ verbose_logger.debug(f"Vertex Agent Engine: Authenticated for project {project_id}")
+
+ return {
+ "Authorization": f"Bearer {access_token}",
+ "Content-Type": "application/json",
+ }
+
+ def _get_user_id(self, optional_params: dict) -> str:
+ """Get or generate user ID for session management."""
+ user_id = optional_params.get("user_id") or optional_params.get("user")
+ if user_id:
+ return user_id
+ # Generate a user ID
+ return f"litellm-user-{str(uuid.uuid4())[:8]}"
+
+ def _get_session_id(self, optional_params: dict) -> Optional[str]:
+ """Get session ID if provided."""
+ return optional_params.get("session_id")
+
+ def transform_request(
+ self,
+ model: str,
+ messages: List[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ headers: dict,
+ ) -> dict:
+ """
+ Transform the request to Vertex Agent Engine format.
+
+ The API expects:
+ {
+ "class_method": "stream_query",
+ "input": {
+ "message": "...",
+ "user_id": "...",
+ "session_id": "..." (optional)
+ }
+ }
+ """
+ # Use the last message content as the prompt
+ prompt = convert_content_list_to_str(messages[-1])
+
+ # Get user_id and session_id
+ user_id = self._get_user_id(optional_params)
+ session_id = self._get_session_id(optional_params)
+
+ # Build the input
+ input_data: Dict[str, Any] = {
+ "message": prompt,
+ "user_id": user_id,
+ }
+
+ if session_id:
+ input_data["session_id"] = session_id
+
+ # Build the request payload
+ # Note: stream_query is used for both streaming and non-streaming
+ # The difference is the endpoint (:streamQuery vs :query)
+ payload = {
+ "class_method": "stream_query",
+ "input": input_data,
+ }
+
+ verbose_logger.debug(f"Vertex Agent Engine payload: {payload}")
+ return payload
+
+ def validate_environment(
+ self,
+ headers: dict,
+ model: str,
+ messages: List[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ api_key: Optional[str] = None,
+ api_base: Optional[str] = None,
+ ) -> dict:
+ """Validate environment and set up authentication headers."""
+ auth_headers = self._get_auth_headers(optional_params, litellm_params)
+ headers.update(auth_headers)
+ return headers
+
+ def _extract_text_from_response(self, response_data: dict) -> str:
+ """Extract text content from the response."""
+ # Try to get from content.parts
+ content = response_data.get("content", {})
+ parts = content.get("parts", [])
+ for part in parts:
+ if "text" in part:
+ return part["text"]
+
+ # Try actions.state_delta
+ actions = response_data.get("actions", {})
+ state_delta = actions.get("state_delta", {})
+ for key, value in state_delta.items():
+ if isinstance(value, str) and value:
+ return value
+
+ return ""
+
+ def _calculate_usage(
+ self, model: str, messages: List[AllMessageValues], content: str
+ ) -> Optional[Usage]:
+ """Calculate token usage using LiteLLM's token counter."""
+ try:
+ from litellm.utils import token_counter
+
+ prompt_tokens = token_counter(model="gpt-3.5-turbo", messages=messages)
+ completion_tokens = token_counter(
+ model="gpt-3.5-turbo", text=content, count_response_tokens=True
+ )
+ total_tokens = prompt_tokens + completion_tokens
+
+ return Usage(
+ prompt_tokens=prompt_tokens,
+ completion_tokens=completion_tokens,
+ total_tokens=total_tokens,
+ )
+ except Exception as e:
+ verbose_logger.warning(f"Failed to calculate token usage: {str(e)}")
+ return None
+
+ def transform_response(
+ self,
+ model: str,
+ raw_response: httpx.Response,
+ model_response: ModelResponse,
+ logging_obj: LiteLLMLoggingObj,
+ request_data: dict,
+ messages: List[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ encoding: Any,
+ api_key: Optional[str] = None,
+ json_mode: Optional[bool] = None,
+ ) -> ModelResponse:
+ """
+ Transform Vertex Agent Engine response to LiteLLM ModelResponse format.
+
+ The response is a streaming SSE format even for non-streaming requests.
+ We need to collect all the chunks and extract the final response.
+ """
+ try:
+ content_type = raw_response.headers.get("content-type", "").lower()
+ verbose_logger.debug(f"Vertex Agent Engine response Content-Type: {content_type}")
+
+ # Parse the SSE response
+ response_text = raw_response.text
+ verbose_logger.debug(f"Response (first 500 chars): {response_text[:500]}")
+
+ # Extract content from SSE stream
+ content = ""
+ for line in response_text.strip().split("\n"):
+ line = line.strip()
+ if not line:
+ continue
+
+ try:
+ data = json.loads(line)
+ if isinstance(data, dict):
+ text = self._extract_text_from_response(data)
+ if text:
+ content = text # Use the last non-empty text
+ except json.JSONDecodeError:
+ continue
+
+ # Create the message
+ message = Message(content=content, role="assistant")
+
+ # Create choices
+ choice = Choices(finish_reason="stop", index=0, message=message)
+
+ # Update model response
+ model_response.choices = [choice]
+ model_response.model = model
+
+ # Calculate usage
+ calculated_usage = self._calculate_usage(model, messages, content)
+ if calculated_usage:
+ setattr(model_response, "usage", calculated_usage)
+
+ return model_response
+
+ except Exception as e:
+ verbose_logger.error(f"Error processing Vertex Agent Engine response: {str(e)}")
+ raise VertexAgentEngineError(
+ message=f"Error processing response: {str(e)}",
+ status_code=raw_response.status_code,
+ )
+
+ def get_streaming_response(
+ self,
+ model: str,
+ raw_response: httpx.Response,
+ ) -> VertexAgentEngineResponseIterator:
+ """Return a streaming iterator for SSE responses."""
+ return VertexAgentEngineResponseIterator(
+ streaming_response=raw_response.iter_lines(),
+ sync_stream=True,
+ )
+
+ def get_sync_custom_stream_wrapper(
+ self,
+ model: str,
+ custom_llm_provider: str,
+ logging_obj: LiteLLMLoggingObj,
+ api_base: str,
+ headers: dict,
+ data: dict,
+ messages: list,
+ client: Optional[Union[HTTPHandler, "AsyncHTTPHandler"]] = None,
+ json_mode: Optional[bool] = None,
+ signed_json_body: Optional[bytes] = None,
+ ) -> "CustomStreamWrapper":
+ """Get a CustomStreamWrapper for synchronous streaming."""
+ from litellm.llms.custom_httpx.http_handler import (
+ HTTPHandler,
+ _get_httpx_client,
+ )
+ from litellm.utils import CustomStreamWrapper
+
+ if client is None or not isinstance(client, HTTPHandler):
+ client = _get_httpx_client(params={})
+
+ # Avoid logging sensitive api_base directly
+ verbose_logger.debug("Making sync streaming request to Vertex AI endpoint.")
+
+ # Make streaming request
+ response = client.post(
+ api_base,
+ headers=headers,
+ data=json.dumps(data),
+ stream=True,
+ logging_obj=logging_obj,
+ )
+
+ if response.status_code != 200:
+ raise VertexAgentEngineError(
+ status_code=response.status_code, message=str(response.read())
+ )
+
+ # Create iterator for SSE stream
+ completion_stream = self.get_streaming_response(model=model, raw_response=response)
+
+ streaming_response = CustomStreamWrapper(
+ completion_stream=completion_stream,
+ model=model,
+ custom_llm_provider=custom_llm_provider,
+ logging_obj=logging_obj,
+ )
+
+ # LOGGING
+ logging_obj.post_call(
+ input=messages,
+ api_key="",
+ original_response="first stream response received",
+ additional_args={"complete_input_dict": data},
+ )
+
+ return streaming_response
+
+ async def get_async_custom_stream_wrapper(
+ self,
+ model: str,
+ custom_llm_provider: str,
+ logging_obj: LiteLLMLoggingObj,
+ api_base: str,
+ headers: dict,
+ data: dict,
+ messages: list,
+ client: Optional["AsyncHTTPHandler"] = None,
+ json_mode: Optional[bool] = None,
+ signed_json_body: Optional[bytes] = None,
+ ) -> "CustomStreamWrapper":
+ """Get a CustomStreamWrapper for asynchronous streaming."""
+ from litellm.llms.custom_httpx.http_handler import (
+ AsyncHTTPHandler,
+ get_async_httpx_client,
+ )
+ from litellm.utils import CustomStreamWrapper
+
+ if client is None or not isinstance(client, AsyncHTTPHandler):
+ client = get_async_httpx_client(
+ llm_provider=cast(Any, "vertex_ai"), params={}
+ )
+
+ # Avoid logging sensitive api_base directly
+ verbose_logger.debug("Making async streaming request to Vertex AI endpoint.")
+
+ # Make async streaming request
+ response = await client.post(
+ api_base,
+ headers=headers,
+ data=json.dumps(data),
+ stream=True,
+ logging_obj=logging_obj,
+ )
+
+ if response.status_code != 200:
+ raise VertexAgentEngineError(
+ status_code=response.status_code, message=str(await response.aread())
+ )
+
+ # Create iterator for SSE stream (async)
+ completion_stream = VertexAgentEngineResponseIterator(
+ streaming_response=response.aiter_lines(),
+ sync_stream=False,
+ )
+
+ streaming_response = CustomStreamWrapper(
+ completion_stream=completion_stream,
+ model=model,
+ custom_llm_provider=custom_llm_provider,
+ logging_obj=logging_obj,
+ )
+
+ # LOGGING
+ logging_obj.post_call(
+ input=messages,
+ api_key="",
+ original_response="first stream response received",
+ additional_args={"complete_input_dict": data},
+ )
+
+ return streaming_response
+
+ @property
+ def has_custom_stream_wrapper(self) -> bool:
+ """Indicates that this config has custom streaming support."""
+ return True
+
+ @property
+ def supports_stream_param_in_request_body(self) -> bool:
+ """Agent Engine does not allow passing `stream` in the request body."""
+ return False
+
+ def get_error_class(
+ self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
+ ) -> BaseLLMException:
+ return VertexAgentEngineError(status_code=status_code, message=error_message)
+
+ def should_fake_stream(
+ self,
+ model: Optional[str],
+ stream: Optional[bool],
+ custom_llm_provider: Optional[str] = None,
+ ) -> bool:
+ """Agent Engine always returns SSE streams, so we use real streaming."""
+ return False
+
diff --git a/litellm/llms/vertex_ai/common_utils.py b/litellm/llms/vertex_ai/common_utils.py
index 3cfa55c0606..6bb11430f20 100644
--- a/litellm/llms/vertex_ai/common_utils.py
+++ b/litellm/llms/vertex_ai/common_utils.py
@@ -5,7 +5,6 @@ from typing import Any, Dict, List, Literal, Optional, Set, Tuple, Union, get_ty
import httpx
import litellm
-from litellm.utils import supports_response_schema, supports_system_messages
from litellm._logging import verbose_logger
from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH
from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs
@@ -14,6 +13,7 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.types.llms.openai import AllMessageValues
from litellm.types.llms.vertex_ai import PartType, Schema
from litellm.types.utils import TokenCountResponse
+from litellm.utils import supports_response_schema, supports_system_messages
class VertexAIError(BaseLLMException):
@@ -36,6 +36,7 @@ class VertexAIModelRoute(str, Enum):
MODEL_GARDEN = "model_garden"
NON_GEMINI = "non_gemini"
OPENAI_COMPATIBLE = "openai"
+ AGENT_ENGINE = "agent_engine"
VERTEX_AI_MODEL_ROUTES = [f"{route.value}/" for route in VertexAIModelRoute]
@@ -76,6 +77,10 @@ def get_vertex_ai_model_route(
if litellm_params and litellm_params.get("base_model") is not None:
if "gemini" in litellm_params["base_model"]:
return VertexAIModelRoute.GEMINI
+
+ # Check for agent_engine models (Reasoning Engines)
+ if "agent_engine/" in model:
+ return VertexAIModelRoute.AGENT_ENGINE
# Check if numeric endpoint ID with custom api_base (PSC endpoint)
# Route to GEMINI (HTTP path) to support PSC endpoints properly
diff --git a/litellm/llms/vertex_ai/image_generation/vertex_gemini_transformation.py b/litellm/llms/vertex_ai/image_generation/vertex_gemini_transformation.py
index b9747652362..619bd006300 100644
--- a/litellm/llms/vertex_ai/image_generation/vertex_gemini_transformation.py
+++ b/litellm/llms/vertex_ai/image_generation/vertex_gemini_transformation.py
@@ -13,7 +13,7 @@ from litellm.types.llms.openai import (
AllMessageValues,
OpenAIImageGenerationOptionalParams,
)
-from litellm.types.utils import ImageObject, ImageResponse
+from litellm.types.utils import ImageObject, ImageResponse, ImageUsage, ImageUsageInputTokensDetails
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
@@ -234,6 +234,27 @@ class VertexAIGeminiImageGenerationConfig(BaseImageGenerationConfig, VertexLLM):
return request_body
+ def _transform_image_usage(self, usage: dict) -> ImageUsage:
+ input_tokens_details = ImageUsageInputTokensDetails(
+ image_tokens=0,
+ text_tokens=0,
+ )
+ tokens_details = usage.get("promptTokensDetails", [])
+ for details in tokens_details:
+ if isinstance(details, dict) and (modality := details.get("modality")):
+ token_count = details.get("tokenCount", 0)
+ if modality == "TEXT":
+ input_tokens_details.text_tokens += token_count
+ elif modality == "IMAGE":
+ input_tokens_details.image_tokens += token_count
+
+ return ImageUsage(
+ input_tokens=usage.get("promptTokenCount", 0),
+ input_tokens_details=input_tokens_details,
+ output_tokens=usage.get("candidatesTokenCount", 0),
+ total_tokens=usage.get("totalTokenCount", 0),
+ )
+
def transform_image_generation_response(
self,
model: str,
@@ -276,6 +297,9 @@ class VertexAIGeminiImageGenerationConfig(BaseImageGenerationConfig, VertexLLM):
b64_json=inline_data["data"],
url=None,
))
+
+ if usage_metadata := response_data.get("usageMetadata", None):
+ model_response.usage = self._transform_image_usage(usage_metadata)
return model_response
diff --git a/litellm/main.py b/litellm/main.py
index 216a3fe0ffd..99788291db2 100644
--- a/litellm/main.py
+++ b/litellm/main.py
@@ -3242,6 +3242,37 @@ def completion( # type: ignore # noqa: PLR0915
timeout=timeout,
client=client,
)
+ elif model_route == VertexAIModelRoute.AGENT_ENGINE:
+ # Vertex AI Agent Engine (Reasoning Engines)
+ from litellm.llms.vertex_ai.agent_engine.transformation import (
+ VertexAgentEngineConfig,
+ )
+
+ vertex_agent_engine_config = VertexAgentEngineConfig()
+
+ # Update litellm_params with vertex credentials
+ litellm_params["vertex_project"] = vertex_ai_project
+ litellm_params["vertex_location"] = vertex_ai_location
+ litellm_params["vertex_credentials"] = vertex_credentials
+
+ model_response = base_llm_http_handler.completion(
+ model=model,
+ stream=stream,
+ messages=messages,
+ model_response=model_response,
+ optional_params=new_params,
+ litellm_params=litellm_params, # type: ignore
+ encoding=encoding,
+ api_key=None,
+ api_base=api_base,
+ logging_obj=logging,
+ acompletion=acompletion,
+ timeout=timeout,
+ client=client,
+ custom_llm_provider="vertex_ai",
+ provider_config=vertex_agent_engine_config,
+ headers=headers or {},
+ )
else: # VertexAIModelRoute.NON_GEMINI
model_response = vertex_ai_non_gemini.completion(
model=model,
@@ -4476,6 +4507,12 @@ def embedding( # noqa: PLR0915
if extra_headers is not None:
optional_params["extra_headers"] = extra_headers
+
+ if encoding_format is not None:
+ optional_params["encoding_format"] = encoding_format
+ else:
+ # Omiting causes openai sdk to add default value of "float"
+ optional_params["encoding_format"] = None
api_version = None
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 26aae425fae..2a7f8aa3ddf 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -30628,11 +30628,11 @@
"litellm_provider": "fireworks_ai",
"mode": "embedding"
},
- "fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b": {
+ "fireworks_ai/accounts/fireworks/models/": {
"max_tokens": 40960,
"max_input_tokens": 40960,
"max_output_tokens": 40960,
- "input_cost_per_token": 0.0,
+ "input_cost_per_token": 1e-07,
"output_cost_per_token": 0.0,
"litellm_provider": "fireworks_ai",
"mode": "embedding"
diff --git a/litellm/proxy/_experimental/out/assets/logos/pydantic.svg b/litellm/proxy/_experimental/out/assets/logos/pydantic.svg
new file mode 100644
index 00000000000..0ff8e5c44c7
--- /dev/null
+++ b/litellm/proxy/_experimental/out/assets/logos/pydantic.svg
@@ -0,0 +1,5 @@
+
diff --git a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py
index fb14ccce50c..62c997659bd 100644
--- a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py
+++ b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py
@@ -605,13 +605,13 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
"""
Only raise exception for "BLOCKED" actions, not for "ANONYMIZED" actions.
- If `self.mask_request_content` or `self.mask_response_content` is set to `True`, then use the output from the guardrail to mask the request or response content.
+ If `self.mask_request_content` or `self.mask_response_content` is set to `True`,
+ then use the output from the guardrail to mask the request or response content.
+
+ However, even with masking enabled, content with action="BLOCKED" should still
+ raise an exception, only content with action="ANONYMIZED" should be masked.
"""
- # if user opted into masking, return False. since we'll use the masked output from the guardrail
- if self.mask_request_content or self.mask_response_content:
- return False
-
# if no intervention, return False
if response.get("action") != "GUARDRAIL_INTERVENED":
return False
diff --git a/litellm/proxy/public_endpoints/agent_create_fields.json b/litellm/proxy/public_endpoints/agent_create_fields.json
index c559b61a76f..931c9a43498 100644
--- a/litellm/proxy/public_endpoints/agent_create_fields.json
+++ b/litellm/proxy/public_endpoints/agent_create_fields.json
@@ -166,6 +166,29 @@
"litellm_params_template": {
"custom_llm_provider": "pydantic_ai_agents"
}
+ },
+ {
+ "agent_type": "vertex_agent_engine",
+ "agent_type_display_name": "Vertex AI Agent Engine",
+ "description": "Connect to Google Cloud Vertex AI Reasoning Engines",
+ "logo_url": "/ui/assets/logos/google.svg",
+ "inherit_credentials_from_provider": "Vertex_AI",
+ "model_template": "vertex_ai/agent_engine/{reasoning_engine_id}",
+ "credential_fields": [
+ {
+ "key": "reasoning_engine_id",
+ "label": "Reasoning Engine Resource ID",
+ "placeholder": "projects/123456789/locations/us-central1/reasoningEngines/987654321",
+ "tooltip": "The full resource ID of your Vertex AI Reasoning Engine. Find this in Google Cloud Console under Vertex AI > Agent Builder > Your Agent.",
+ "required": true,
+ "field_type": "text",
+ "default_value": null,
+ "include_in_litellm_params": false
+ }
+ ],
+ "litellm_params_template": {
+ "custom_llm_provider": "vertex_ai"
+ }
}
]
diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json
index 629760a7dd2..68264a576fe 100644
--- a/litellm/proxy/public_endpoints/provider_create_fields.json
+++ b/litellm/proxy/public_endpoints/provider_create_fields.json
@@ -2689,8 +2689,8 @@
"key": "vertex_credentials",
"label": "Vertex Credentials",
"placeholder": null,
- "tooltip": null,
- "required": true,
+ "tooltip": "Optional - Upload your GCP service account JSON file. If not provided, uses default GCP credentials (ADC).",
+ "required": false,
"field_type": "upload",
"options": null,
"default_value": null
diff --git a/litellm/utils.py b/litellm/utils.py
index 524e86cfbbe..169a2cdcb1a 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -1,3 +1,6 @@
+# from __future__ import annotations must be the first non-comment statement
+from __future__ import annotations
+
# +-----------------------------------------------+
# | |
# | Give Feedback / Get Help |
@@ -96,11 +99,11 @@ from litellm.litellm_core_utils.core_helpers import (
process_response_headers,
)
from litellm.litellm_core_utils.credential_accessor import CredentialAccessor
-from litellm.litellm_core_utils.default_encoding import encoding
from litellm.litellm_core_utils.dot_notation_indexing import (
delete_nested_value,
is_nested_path,
)
+from litellm._lazy_imports import _get_default_encoding
from litellm.litellm_core_utils.exception_mapping_utils import (
_get_response_headers,
exception_type,
@@ -260,12 +263,16 @@ from litellm.llms.base_llm.base_utils import (
BaseLLMModelInfo,
type_to_response_format_param,
)
+
+if TYPE_CHECKING:
+ # Heavy types that are only needed for type checking; avoid importing
+ # their modules at runtime during `litellm` import.
+ from litellm.llms.base_llm.files.transformation import BaseFilesConfig
from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig
from litellm.llms.base_llm.chat.transformation import BaseConfig
from litellm.llms.base_llm.completion.transformation import BaseTextCompletionConfig
from litellm.llms.base_llm.containers.transformation import BaseContainerConfig
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
-from litellm.llms.base_llm.files.transformation import BaseFilesConfig
from litellm.llms.base_llm.image_edit.transformation import BaseImageEditConfig
from litellm.llms.base_llm.image_generation.transformation import (
BaseImageGenerationConfig,
@@ -293,6 +300,7 @@ from .caching.caching import (
RedisSemanticCache,
S3Cache,
)
+
from .exceptions import (
APIConnectionError,
APIError,
@@ -1752,7 +1760,7 @@ def _select_tokenizer_helper(model: str) -> SelectTokenizerResponse:
def _return_openai_tokenizer(model: str) -> SelectTokenizerResponse:
- return {"type": "openai_tokenizer", "tokenizer": encoding}
+ return {"type": "openai_tokenizer", "tokenizer": _get_default_encoding()}
def _return_huggingface_tokenizer(model: str) -> Optional[SelectTokenizerResponse]:
@@ -5842,7 +5850,7 @@ def prompt_token_calculator(model, messages):
anthropic_obj = Anthropic()
num_tokens = anthropic_obj.count_tokens(text) # type: ignore
else:
- num_tokens = len(encoding.encode(text))
+ num_tokens = len(_get_default_encoding().encode(text))
return num_tokens
@@ -8187,9 +8195,6 @@ def extract_duration_from_srt_or_vtt(srt_or_vtt_content: str) -> Optional[float]
return max(durations) if durations else None
-import httpx
-
-
def _add_path_to_api_base(api_base: str, ending_path: str) -> str:
"""
Adds an ending path to an API base URL while preventing duplicate path segments.
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index 26aae425fae..2a7f8aa3ddf 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -30628,11 +30628,11 @@
"litellm_provider": "fireworks_ai",
"mode": "embedding"
},
- "fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b": {
+ "fireworks_ai/accounts/fireworks/models/": {
"max_tokens": 40960,
"max_input_tokens": 40960,
"max_output_tokens": 40960,
- "input_cost_per_token": 0.0,
+ "input_cost_per_token": 1e-07,
"output_cost_per_token": 0.0,
"litellm_provider": "fireworks_ai",
"mode": "embedding"
diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json
index cec9f9e37c0..d63b26d55fe 100644
--- a/provider_endpoints_support.json
+++ b/provider_endpoints_support.json
@@ -1946,6 +1946,40 @@
"rerank": false,
"a2a": true
}
+ },
+ "vertex_ai/agent_engine": {
+ "display_name": "Vertex AI Agent Engine (`vertex_ai/agent_engine`)",
+ "url": "https://docs.litellm.ai/docs/providers/vertex_ai_agent_engine",
+ "endpoints": {
+ "chat_completions": true,
+ "messages": true,
+ "responses": true,
+ "embeddings": false,
+ "image_generations": false,
+ "audio_transcriptions": false,
+ "audio_speech": false,
+ "moderations": false,
+ "batches": false,
+ "rerank": false,
+ "a2a": true
+ }
+ },
+ "pydantic_ai_agents": {
+ "display_name": "Pydantic AI Agents (`pydantic_ai_agents`)",
+ "url": "https://docs.litellm.ai/docs/providers/pydantic_ai_agent",
+ "endpoints": {
+ "chat_completions": false,
+ "messages": false,
+ "responses": false,
+ "embeddings": false,
+ "image_generations": false,
+ "audio_transcriptions": false,
+ "audio_speech": false,
+ "moderations": false,
+ "batches": false,
+ "rerank": false,
+ "a2a": true
+ }
}
}
}
\ No newline at end of file
diff --git a/proxy_server_config.yaml b/proxy_server_config.yaml
index df3a08a143b..deb71225390 100644
--- a/proxy_server_config.yaml
+++ b/proxy_server_config.yaml
@@ -152,6 +152,7 @@ model_list:
litellm_settings:
# set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
drop_params: True
+ success_callback: ["prometheus"]
# max_budget: 100
# budget_duration: 30d
num_retries: 5
diff --git a/requirements.txt b/requirements.txt
index 3e64600fc64..69eaaff0835 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -13,6 +13,7 @@ uvloop==0.21.0 # uvicorn dep, gives us much better performance under load
boto3==1.36.0 # aws bedrock/sagemaker calls
redis==5.2.1 # redis caching
prisma==0.11.0 # for db
+nodejs-bin==18.4.0a4 ## required by prisma for migrations, prevents runtime download
mangum==0.17.0 # for aws lambda functions
pynacl==1.5.0 # for encrypting keys
google-cloud-aiplatform==1.47.0 # for vertex ai calls
diff --git a/tests/agent_tests/local_vertex_agent.py b/tests/agent_tests/local_vertex_agent.py
new file mode 100644
index 00000000000..3cc9f868612
--- /dev/null
+++ b/tests/agent_tests/local_vertex_agent.py
@@ -0,0 +1,151 @@
+"""
+Test script for Vertex AI Reasoning Engine.
+
+This script demonstrates how to:
+1. Authenticate with Google Cloud
+2. Send queries to a Vertex AI Reasoning Engine using the :query endpoint
+
+Usage:
+ python local_vertex_agent.py
+
+Requirements:
+ pip install httpx google-auth
+"""
+
+import asyncio
+import json
+from uuid import uuid4
+
+from google.auth import default
+from google.auth.transport.requests import Request
+import httpx
+
+# Configuration - update these for your agent
+PROJECT_ID = "gen-lang-client-0682925754" # Your GCP project ID
+LOCATION = "us-central1" # Your agent's location
+
+# For Reasoning Engines, use just the numeric ID at the end
+REASONING_ENGINE_ID = "8263861224643493888"
+
+# The project number from the resource name
+PROJECT_NUMBER = "1060139831167"
+
+
+async def main():
+ """Main function to test Vertex AI Reasoning Engine."""
+
+ # Step 1: Authenticate with Google Cloud
+ print("Step 1: Authenticating with Google Cloud...")
+ credentials, project = default(scopes=['https://www.googleapis.com/auth/cloud-platform'])
+ credentials.refresh(Request())
+ print(f"Authenticated! Project: {project}")
+ print(f"Token (first 20 chars): {credentials.token[:20]}...")
+
+ # Step 2: Build the endpoint URL
+ base_url = f"https://{LOCATION}-aiplatform.googleapis.com"
+ resource_path = f"projects/{PROJECT_NUMBER}/locations/{LOCATION}/reasoningEngines/{REASONING_ENGINE_ID}"
+
+ # The Reasoning Engine uses :query endpoint with specific format
+ query_url = f"{base_url}/v1beta1/{resource_path}:query"
+ stream_url = f"{base_url}/v1beta1/{resource_path}:streamQuery"
+
+ print(f"\nQuery URL: {query_url}")
+ print(f"Stream URL: {stream_url}")
+
+ # Step 3: Create authenticated httpx client
+ print("\nStep 2: Creating authenticated HTTP client...")
+ client = httpx.AsyncClient(
+ headers={
+ "Authorization": f"Bearer {credentials.token}",
+ "Content-Type": "application/json",
+ },
+ timeout=120.0,
+ )
+
+ # Step 4: Build the query request (non-streaming)
+ # Note: For non-streaming, we need to:
+ # 1. Create a session
+ # 2. Use the streaming endpoint with stream_query method
+ # The :query endpoint only supports session management methods
+
+ user_id = f"test-user-{uuid4().hex[:8]}"
+
+ # First create a session
+ create_session_request = {
+ "class_method": "async_create_session",
+ "input": {
+ "user_id": user_id,
+ }
+ }
+
+ print(f"\nStep 3: Creating session...")
+ print(f"User ID: {user_id}")
+
+ async with client:
+ # Create session
+ print(f"\nSending to: {query_url}")
+ response = await client.post(query_url, json=create_session_request)
+ print(f"Create session status: {response.status_code}")
+
+ if response.status_code == 200:
+ session_data = response.json()
+ print(f"Session created:\n{json.dumps(session_data, indent=2)}")
+
+ # Extract session_id from response
+ session_id = session_data.get("output", {}).get("id") or session_data.get("output", {}).get("session_id")
+ print(f"\nSession ID: {session_id}")
+
+ # Now send the actual query via streamQuery
+ query_request = {
+ "class_method": "stream_query",
+ "input": {
+ "message": "Hello! What can you do?",
+ "user_id": user_id,
+ "session_id": session_id,
+ }
+ }
+
+ print(f"\nStep 4: Sending query via streamQuery...")
+ print(f"Request:\n{json.dumps(query_request, indent=2)}")
+
+ # Use streaming endpoint but collect full response
+ async with client.stream("POST", stream_url, json=query_request) as stream_response:
+ print(f"Query status: {stream_response.status_code}")
+
+ if stream_response.status_code == 200:
+ print("\nResponse:")
+ full_response = ""
+ async for line in stream_response.aiter_lines():
+ if line:
+ full_response = line # Keep last line (full response)
+
+ # Parse and display
+ try:
+ data = json.loads(full_response)
+ # Extract the text from the response
+ content = data.get("content", {})
+ parts = content.get("parts", [])
+ for part in parts:
+ if "text" in part:
+ print(f"\nAgent response:\n{part['text']}")
+ except:
+ print(full_response)
+ else:
+ content = await stream_response.aread()
+ print(f"Error: {content.decode()}")
+ else:
+ print(f"Error creating session: {response.text}")
+
+
+if __name__ == "__main__":
+ print("=" * 60)
+ print("Vertex AI Reasoning Engine Test Script")
+ print("=" * 60)
+ print(f"\nConfiguration:")
+ print(f" PROJECT_ID: {PROJECT_ID}")
+ print(f" PROJECT_NUMBER: {PROJECT_NUMBER}")
+ print(f" LOCATION: {LOCATION}")
+ print(f" REASONING_ENGINE_ID: {REASONING_ENGINE_ID}")
+ print()
+
+ asyncio.run(main())
diff --git a/tests/agent_tests/test_a2a_completion_bridge.py b/tests/agent_tests/test_a2a_completion_bridge.py
index 4191821f3de..224809dd7f5 100644
--- a/tests/agent_tests/test_a2a_completion_bridge.py
+++ b/tests/agent_tests/test_a2a_completion_bridge.py
@@ -201,3 +201,79 @@ async def test_a2a_completion_bridge_bedrock_agentcore():
print(f"Received {len(chunks)} chunks from Bedrock AgentCore")
+
+# ============================================================
+# Vertex AI Agent Engine Tests
+# ============================================================
+
+# Configuration - update these for your Vertex AI Reasoning Engine
+VERTEX_AGENT_RESOURCE_NAME = "projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888"
+
+
+@pytest.mark.asyncio
+async def test_vertex_agent_engine_non_streaming():
+ """
+ Test non-streaming request to Vertex AI Agent Engine via litellm.acompletion.
+
+ Uses the Reasoning Engine resource ID to call a hosted agent.
+ """
+
+ litellm._turn_on_debug()
+
+ # Call via litellm.acompletion with vertex_ai/agent_engine/ prefix
+ response = await litellm.acompletion(
+ model=f"vertex_ai/agent_engine/{VERTEX_AGENT_RESOURCE_NAME}",
+ messages=[{"role": "user", "content": "Hello! What can you do?"}],
+ stream=False,
+ )
+
+ print(f"\n=== Vertex Agent Engine Non-Streaming Response ===")
+ print(f"Response: {response}")
+
+ # Basic assertions
+ assert response is not None
+ assert hasattr(response, "choices")
+ assert len(response.choices) > 0
+ assert response.choices[0].message is not None
+ assert response.choices[0].message.content is not None
+ assert len(response.choices[0].message.content) > 0
+
+ print(f"Agent response: {response.choices[0].message.content[:200]}...")
+
+
+@pytest.mark.asyncio
+async def test_vertex_agent_engine_streaming():
+ """
+ Test streaming request to Vertex AI Agent Engine via litellm.acompletion.
+
+ Uses the Reasoning Engine resource ID to call a hosted agent with streaming.
+ """
+ #litellm._turn_on_debug()
+
+ # Call via litellm.acompletion with streaming
+ response = await litellm.acompletion(
+ model=f"vertex_ai/agent_engine/{VERTEX_AGENT_RESOURCE_NAME}",
+ messages=[{"role": "user", "content": "Hello! What can you do?"}],
+ stream=True,
+ )
+
+ print(f"\n=== Vertex Agent Engine Streaming Response ===")
+
+ chunks = []
+ full_content = ""
+ async for chunk in response:
+ print(f"Chunk: {chunk}")
+ # chunks.append(chunk)
+ # if hasattr(chunk, "choices") and len(chunk.choices) > 0:
+ # delta = chunk.choices[0].delta
+ # if hasattr(delta, "content") and delta.content:
+ # full_content += delta.content
+ # print(f"Chunk: {delta.content}", end="", flush=True)
+
+ # # print(f"\n\nReceived {len(chunks)} chunks")
+ # print(f"Full content: {full_content[:200]}...")
+
+ # # Basic assertions
+ # assert len(chunks) > 0
+ # assert len(full_content) > 0
+
diff --git a/tests/litellm/llms/vertex_ai/agent_engine/test_transformation.py b/tests/litellm/llms/vertex_ai/agent_engine/test_transformation.py
new file mode 100644
index 00000000000..cb3a5807d8c
--- /dev/null
+++ b/tests/litellm/llms/vertex_ai/agent_engine/test_transformation.py
@@ -0,0 +1,128 @@
+"""
+Tests for Vertex AI Agent Engine transformation.
+
+Tests the request transformation and streaming chunk parsing without making real API calls.
+"""
+
+import os
+import sys
+
+import pytest
+
+sys.path.insert(0, os.path.abspath("../../../../.."))
+
+from litellm.llms.vertex_ai.agent_engine.sse_iterator import (
+ VertexAgentEngineResponseIterator,
+)
+from litellm.llms.vertex_ai.agent_engine.transformation import VertexAgentEngineConfig
+
+
+class TestVertexAgentEngineTransformRequest:
+ """Tests for transform_request method."""
+
+ def test_transform_request_basic(self):
+ """
+ Test that transform_request correctly formats messages into Vertex Agent Engine payload.
+ """
+ config = VertexAgentEngineConfig()
+
+ messages = [{"role": "user", "content": "Hello, what can you do?"}]
+ optional_params = {"user_id": "test-user-123"}
+ litellm_params = {}
+
+ result = config.transform_request(
+ model="agent_engine/123456789",
+ messages=messages,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ headers={},
+ )
+
+ assert result["class_method"] == "stream_query"
+ assert result["input"]["message"] == "Hello, what can you do?"
+ assert result["input"]["user_id"] == "test-user-123"
+ assert "session_id" not in result["input"]
+
+ def test_transform_request_with_session_id(self):
+ """
+ Test that transform_request includes session_id when provided.
+ """
+ config = VertexAgentEngineConfig()
+
+ messages = [{"role": "user", "content": "Follow up question"}]
+ optional_params = {
+ "user_id": "test-user-123",
+ "session_id": "session-abc-456",
+ }
+ litellm_params = {}
+
+ result = config.transform_request(
+ model="agent_engine/123456789",
+ messages=messages,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ headers={},
+ )
+
+ assert result["class_method"] == "stream_query"
+ assert result["input"]["message"] == "Follow up question"
+ assert result["input"]["user_id"] == "test-user-123"
+ assert result["input"]["session_id"] == "session-abc-456"
+
+
+class TestVertexAgentEngineChunkParser:
+ """Tests for the streaming chunk parser."""
+
+ def test_chunk_parser_with_text_content(self):
+ """
+ Test that chunk_parser correctly extracts text from Vertex Agent Engine response format.
+ """
+ iterator = VertexAgentEngineResponseIterator(
+ streaming_response=iter([]),
+ sync_stream=True,
+ )
+
+ chunk = {
+ "content": {
+ "parts": [{"text": "Hello! I can help you with financial analysis."}],
+ "role": "model",
+ },
+ "finish_reason": "STOP",
+ "usage_metadata": {
+ "prompt_token_count": 100,
+ "candidates_token_count": 50,
+ "total_token_count": 150,
+ },
+ }
+
+ result = iterator.chunk_parser(chunk)
+
+ assert result.choices[0].delta.content == "Hello! I can help you with financial analysis."
+ assert result.choices[0].delta.role == "assistant"
+ assert result.choices[0].finish_reason == "stop"
+ assert result.usage["prompt_tokens"] == 100
+ assert result.usage["completion_tokens"] == 50
+ assert result.usage["total_tokens"] == 150
+
+ def test_chunk_parser_without_finish_reason(self):
+ """
+ Test that chunk_parser handles chunks without finish_reason (intermediate chunks).
+ """
+ iterator = VertexAgentEngineResponseIterator(
+ streaming_response=iter([]),
+ sync_stream=True,
+ )
+
+ chunk = {
+ "content": {
+ "parts": [{"text": "Partial response..."}],
+ "role": "model",
+ },
+ }
+
+ result = iterator.chunk_parser(chunk)
+
+ assert result.choices[0].delta.content == "Partial response..."
+ assert result.choices[0].finish_reason is None
+ assert result.usage is None
+
diff --git a/tests/llm_responses_api_testing/test_azure_responses_api.py b/tests/llm_responses_api_testing/test_azure_responses_api.py
index eeb7eb50151..86b490994d6 100644
--- a/tests/llm_responses_api_testing/test_azure_responses_api.py
+++ b/tests/llm_responses_api_testing/test_azure_responses_api.py
@@ -182,3 +182,94 @@ async def test_azure_responses_api_status_error():
f"Expected: {json.dumps(expected_input, indent=2)}\n"
f"Got: {json.dumps(captured_request_body['input'], indent=2)}"
)
+
+
+@pytest.mark.asyncio
+async def test_azure_responses_api_headers_with_llm_provider_prefix():
+ """
+ Test that Azure-specific headers like 'x-request-id' and 'apim-request-id'
+ are properly forwarded with 'llm_provider-' prefix in response._hidden_params["headers"].
+
+ Issue: https://github.com/BerriAI/litellm/issues/16538
+
+ The fix ensures that processed headers (with llm_provider- prefix) are stored
+ in response._hidden_params["headers"] instead of additional_headers, making them
+ accessible via completion.headers in the same way as the completion API.
+ """
+ import json
+ import httpx
+
+ mock_response_data = {
+ "id": "resp_123",
+ "object": "response",
+ "created_at": 1234567890,
+ "model": "gpt-5-codex",
+ "status": "completed",
+ "output": [
+ {
+ "id": "msg_123",
+ "role": "assistant",
+ "type": "message",
+ "content": [{"type": "output_text", "text": "Hello!"}],
+ }
+ ],
+ }
+
+ # Mock headers that Azure returns - exactly like in the issue
+ mock_headers = {
+ "date": "Wed, 12 Nov 2025 15:31:28 GMT",
+ "server": "uvicorn",
+ "content-type": "application/json",
+ "x-ratelimit-remaining-tokens": "5010000",
+ "x-ratelimit-limit-tokens": "5010000",
+ # These are the Azure-specific headers that should be forwarded with llm_provider- prefix
+ "x-request-id": "12086715-aca3-4006-a29f-2f1e1d552043",
+ "apim-request-id": "25664b0d-cf4b-4e10-8d27-c7272e7efd49",
+ "x-ms-region": "Sweden Central",
+ }
+
+ async def mock_post(*args, **kwargs):
+ response_content = json.dumps(mock_response_data).encode("utf-8")
+ response = httpx.Response(
+ status_code=200,
+ headers=mock_headers,
+ content=response_content,
+ request=httpx.Request(method="POST", url="https://test.openai.azure.com"),
+ )
+ return response
+
+ with patch.object(AsyncHTTPHandler, "post", new=mock_post):
+ response = await litellm.aresponses(
+ model="azure/gpt-5-codex",
+ api_version="2025-03-01-preview",
+ api_base="https://test.openai.azure.com",
+ api_key="test-key",
+ input="Hello, can you tell me a short joke?",
+ )
+
+ # Check that the response has the expected headers structure
+ assert hasattr(response, "_hidden_params"), "Response should have _hidden_params"
+ assert "additional_headers" in response._hidden_params, (
+ "Response _hidden_params should contain 'additional_headers' with the LLM provider headers"
+ )
+
+ headers = response._hidden_params["additional_headers"]
+
+ # Verify that Azure-specific headers are present with llm_provider- prefix
+ assert "llm_provider-x-request-id" in headers, (
+ f"Response should contain 'llm_provider-x-request-id' header. "
+ f"Headers: {list(headers.keys())}"
+ )
+ assert "llm_provider-apim-request-id" in headers, (
+ f"Response should contain 'llm_provider-apim-request-id' header. "
+ f"Headers: {list(headers.keys())}"
+ )
+
+ # Verify the header values match
+ assert headers["llm_provider-x-request-id"] == "12086715-aca3-4006-a29f-2f1e1d552043"
+ assert headers["llm_provider-apim-request-id"] == "25664b0d-cf4b-4e10-8d27-c7272e7efd49"
+ assert headers["llm_provider-x-ms-region"] == "Sweden Central"
+
+ # Also verify openai-compatible headers are included
+ assert "x-ratelimit-limit-tokens" in headers
+ assert "x-ratelimit-remaining-tokens" in headers
diff --git a/tests/local_testing/test_embedding.py b/tests/local_testing/test_embedding.py
index 13ff81bc695..4855932ca9f 100644
--- a/tests/local_testing/test_embedding.py
+++ b/tests/local_testing/test_embedding.py
@@ -1308,3 +1308,110 @@ def test_jina_ai_img_embeddings(input_data, expected_payload_input):
# Assert that the 'input' field in the payload matches our expectation.
assert "input" in sent_data
assert sent_data["input"] == expected_payload_input
+
+
+def test_encoding_format_none_not_omitted_from_openai_sdk():
+ """
+ Test that encoding_format=None is explicitly sent to OpenAI SDK.
+
+ This test verifies that when encoding_format is not provided by the user,
+ liteLLM explicitly sets it to None rather than omitting it. This prevents
+ the OpenAI SDK from adding its default value of 'base64'.
+
+ Without this fix:
+ - OpenAI SDK adds encoding_format='base64' as default when parameter is missing
+ - This causes issues with providers that don't support encoding_format (like Gemini)
+
+ With this fix:
+ - encoding_format=None is explicitly passed
+ - OpenAI SDK respects the explicit None and doesn't add defaults
+ """
+ with patch("litellm.llms.openai.openai.OpenAIChatCompletion._get_openai_client") as mock_get_client:
+ # Create a mock client instance
+ mock_client_instance = MagicMock()
+ mock_get_client.return_value = mock_client_instance
+
+ # Mock the embeddings.with_raw_response.create method
+ mock_response = MagicMock()
+ mock_response.parse.return_value = MagicMock(
+ model_dump=lambda: {
+ 'data': [{'embedding': [0.1, 0.2, 0.3], 'index': 0}],
+ 'model': 'text-embedding-ada-002',
+ 'object': 'list',
+ 'usage': {'prompt_tokens': 1, 'total_tokens': 1}
+ }
+ )
+ mock_response.headers = {}
+
+ mock_client_instance.embeddings.with_raw_response.create.return_value = mock_response
+
+ # Call the embedding function without encoding_format
+ response = embedding(
+ model="text-embedding-ada-002",
+ input="Hello world",
+ )
+
+ # Get the call arguments to verify what was sent to OpenAI SDK
+ call_args = mock_client_instance.embeddings.with_raw_response.create.call_args
+ assert call_args is not None, "OpenAI SDK embeddings.create should have been called"
+
+ call_kwargs = call_args[1] # Get kwargs
+
+ # The key assertion: encoding_format should be in the request with value None
+ # This prevents OpenAI SDK from adding its default 'base64' value
+ assert 'encoding_format' in call_kwargs, (
+ "encoding_format should be explicitly passed to OpenAI SDK "
+ "(even if None) to prevent SDK from adding default value"
+ )
+ assert call_kwargs['encoding_format'] is None, (
+ "encoding_format should be None when not provided by user"
+ )
+
+ print("✅ PASS: encoding_format=None is correctly passed to OpenAI SDK")
+
+
+def test_encoding_format_explicit_value_preserved():
+ """
+ Test that explicitly provided encoding_format values are preserved.
+
+ When user provides encoding_format='float' or 'base64', it should be
+ sent as-is to the OpenAI SDK.
+ """
+ with patch("litellm.llms.openai.openai.OpenAIChatCompletion._get_openai_client") as mock_get_client:
+ # Create a mock client instance
+ mock_client_instance = MagicMock()
+ mock_get_client.return_value = mock_client_instance
+
+ # Mock the embeddings.with_raw_response.create method
+ mock_response = MagicMock()
+ mock_response.parse.return_value = MagicMock(
+ model_dump=lambda: {
+ 'data': [{'embedding': [0.1, 0.2, 0.3], 'index': 0}],
+ 'model': 'text-embedding-ada-002',
+ 'object': 'list',
+ 'usage': {'prompt_tokens': 1, 'total_tokens': 1}
+ }
+ )
+ mock_response.headers = {}
+
+ mock_client_instance.embeddings.with_raw_response.create.return_value = mock_response
+
+ # Test with explicit encoding_format='float'
+ response = embedding(
+ model="text-embedding-ada-002",
+ input="Hello world",
+ encoding_format="float"
+ )
+
+ # Verify the encoding_format was passed correctly
+ call_args = mock_client_instance.embeddings.with_raw_response.create.call_args
+ call_kwargs = call_args[1]
+
+ assert 'encoding_format' in call_kwargs, (
+ "encoding_format should be in the request"
+ )
+ assert call_kwargs['encoding_format'] == 'float', (
+ "encoding_format should be 'float' when explicitly provided"
+ )
+
+ print("✅ PASS: encoding_format='float' is correctly preserved")
diff --git a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
index 0d21c163761..a4da4ebb683 100644
--- a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
+++ b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
@@ -79,3 +79,102 @@ def test_chunk_parser_usage_transformation():
assert "usage" in parsed
assert parsed["usage"]["input_tokens"] == 10
assert parsed["usage"]["output_tokens"] == 5
+
+
+def test_remove_ttl_from_cache_control():
+ """Ensure ttl field is removed from cache_control in messages."""
+
+ cfg = AmazonAnthropicClaudeMessagesConfig()
+
+ # Test case 1: Message with cache_control containing ttl
+ request = {
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Hello",
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "1h"
+ }
+ }
+ ]
+ }
+ ]
+ }
+
+ cfg._remove_ttl_from_cache_control(request)
+
+ # Verify ttl is removed but cache_control remains
+ assert "cache_control" in request["messages"][0]["content"][0]
+ assert "ttl" not in request["messages"][0]["content"][0]["cache_control"]
+ assert request["messages"][0]["content"][0]["cache_control"]["type"] == "ephemeral"
+
+ # Test case 2: Message with multiple content items
+ request2 = {
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Hello",
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "1h"
+ }
+ },
+ {
+ "type": "text",
+ "text": "World",
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "2h"
+ }
+ }
+ ]
+ }
+ ]
+ }
+
+ cfg._remove_ttl_from_cache_control(request2)
+
+ # Verify ttl is removed from all items
+ for item in request2["messages"][0]["content"]:
+ if "cache_control" in item:
+ assert "ttl" not in item["cache_control"]
+
+ # Test case 3: Message without ttl (should remain unchanged)
+ request3 = {
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Hello",
+ "cache_control": {
+ "type": "ephemeral"
+ }
+ }
+ ]
+ }
+ ]
+ }
+
+ cfg._remove_ttl_from_cache_control(request3)
+
+ # Verify cache_control is unchanged
+ assert request3["messages"][0]["content"][0]["cache_control"]["type"] == "ephemeral"
+
+ # Test case 4: Empty messages (should not raise error)
+ request4 = {"messages": []}
+ cfg._remove_ttl_from_cache_control(request4)
+ assert request4 == {"messages": []}
+
+ # Test case 5: Request without messages key (should not raise error)
+ request5 = {}
+ cfg._remove_ttl_from_cache_control(request5)
+ assert request5 == {}
diff --git a/tests/test_litellm/llms/vertex_ai/image_generation/test_vertex_ai_image_generation_transformation.py b/tests/test_litellm/llms/vertex_ai/image_generation/test_vertex_ai_image_generation_transformation.py
index 7cba03c38c8..b91438b3cac 100644
--- a/tests/test_litellm/llms/vertex_ai/image_generation/test_vertex_ai_image_generation_transformation.py
+++ b/tests/test_litellm/llms/vertex_ai/image_generation/test_vertex_ai_image_generation_transformation.py
@@ -141,7 +141,22 @@ class TestVertexAIGeminiImageGenerationConfig:
]
}
}
- ]
+ ],
+ "usageMetadata": {
+ "promptTokenCount": 93,
+ "promptTokensDetails": [
+ {
+ "modality": "TEXT",
+ "tokenCount": 54,
+ },
+ {
+ "modality": "IMAGE",
+ "tokenCount": 39,
+ }
+ ],
+ "candidatesTokenCount": 17,
+ "totalTokenCount": 110,
+ }
}
mock_response.headers = {}
@@ -162,6 +177,12 @@ class TestVertexAIGeminiImageGenerationConfig:
assert len(result.data) == 1
assert result.data[0].b64_json == "base64_encoded_image_data"
assert result.data[0].url is None
+ assert result.usage.input_tokens == 93
+ assert result.usage.input_tokens_details.text_tokens == 54
+ assert result.usage.input_tokens_details.image_tokens == 39
+ assert result.usage.output_tokens == 17
+ assert result.usage.total_tokens == 110
+
def test_transform_image_generation_response_multiple_images(self):
"""Test response transformation with multiple images"""
diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py
index b129b7bab7f..5f2dd387b95 100644
--- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py
+++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py
@@ -72,3 +72,46 @@ def test_vertex_ai_anthropic_web_search_header_in_completion():
# because Anthropic doesn't require it
assert "anthropic-beta" not in headers_non_vertex or "web-search" not in headers_non_vertex.get("anthropic-beta", ""), \
"anthropic-beta with web-search should not be present for non-Vertex requests"
+
+
+def test_vertex_ai_anthropic_structured_output_header_not_added():
+ """Test that structured output beta headers are NOT added for Vertex AI requests"""
+ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
+
+ config = AnthropicConfig()
+
+ # Test case 1: Vertex request with output_format should NOT add beta header
+ headers_vertex = {}
+ optional_params_vertex = {
+ 'output_format': {
+ 'type': 'json_schema',
+ 'json_schema': {
+ 'name': 'MathResult',
+ 'schema': {'properties': {'result': {'type': 'integer'}}}
+ }
+ },
+ 'is_vertex_request': True
+ }
+ result_vertex = config.update_headers_with_optional_anthropic_beta(headers_vertex, optional_params_vertex)
+
+ assert "anthropic-beta" not in result_vertex, \
+ f"Vertex request should NOT have anthropic-beta header for structured output, got: {result_vertex.get('anthropic-beta')}"
+
+ # Test case 2: Non-Vertex request with output_format SHOULD add beta header
+ headers_non_vertex = {}
+ optional_params_non_vertex = {
+ 'output_format': {
+ 'type': 'json_schema',
+ 'json_schema': {
+ 'name': 'MathResult',
+ 'schema': {'properties': {'result': {'type': 'integer'}}}
+ }
+ },
+ 'is_vertex_request': False
+ }
+ result_non_vertex = config.update_headers_with_optional_anthropic_beta(headers_non_vertex, optional_params_non_vertex)
+
+ assert "anthropic-beta" in result_non_vertex, \
+ "Non-Vertex request SHOULD have anthropic-beta header for structured output"
+ assert result_non_vertex["anthropic-beta"] == "structured-outputs-2025-11-13", \
+ f"Expected 'structured-outputs-2025-11-13', got: {result_non_vertex.get('anthropic-beta')}"
diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py
index 69b0bb27b4b..84d320a0a27 100644
--- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py
+++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py
@@ -1101,3 +1101,91 @@ async def test_bedrock_apply_guardrail_with_only_tool_calls_response():
# Verify that the Bedrock API was NOT called since there's no text to process
mock_api_request.assert_not_called()
print("✅ apply_guardrail with tool_calls test passed - no API call made")
+
+
+@pytest.mark.asyncio
+async def test_bedrock_guardrail_blocked_content_with_masking_enabled():
+ """Test that BLOCKED content raises exception even when masking is enabled
+
+ This test verifies the bug fix where previously mask_request_content=True or
+ mask_response_content=True would bypass all BLOCKED content checks. Now it
+ properly distinguishes between BLOCKED (raise exception) and ANONYMIZED (apply masking).
+ """
+
+ # Create guardrail with masking enabled
+ guardrail = BedrockGuardrail(
+ guardrailIdentifier="test-guardrail",
+ guardrailVersion="DRAFT",
+ mask_request_content=True, # Masking enabled
+ mask_response_content=True, # Masking enabled
+ )
+
+ # Mock Bedrock response with BLOCKED content (hate speech)
+ blocked_response = {
+ "action": "GUARDRAIL_INTERVENED",
+ "assessments": [
+ {
+ "contentPolicy": {
+ "filters": [
+ {
+ "type": "HATE",
+ "confidence": "HIGH",
+ "action": "BLOCKED", # Should raise exception
+ }
+ ]
+ },
+ "sensitiveInformationPolicy": {
+ "piiEntities": [
+ {
+ "type": "NAME",
+ "match": "John Doe",
+ "action": "ANONYMIZED", # Should be masked
+ }
+ ]
+ },
+ }
+ ],
+ "outputs": [{"text": "Content blocked due to policy violation"}],
+ }
+
+ mock_bedrock_response = MagicMock()
+ mock_bedrock_response.status_code = 200
+ mock_bedrock_response.json.return_value = blocked_response
+
+ # Mock credentials
+ mock_credentials = MagicMock()
+ mock_credentials.access_key = "test-access-key"
+ mock_credentials.secret_key = "test-secret-key"
+ mock_credentials.token = None
+
+ request_data = {
+ "model": "gpt-4o",
+ "messages": [
+ {"role": "user", "content": "Test message with PII and hate speech"},
+ ],
+ }
+
+ # Mock AWS-related methods
+ with patch.object(
+ guardrail.async_handler, "post", new_callable=AsyncMock
+ ) as mock_post, patch.object(
+ guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1")
+ ), patch.object(
+ guardrail, "_prepare_request", return_value=MagicMock()
+ ):
+ mock_post.return_value = mock_bedrock_response
+
+ # Should raise HTTPException for BLOCKED content
+ with pytest.raises(HTTPException) as exc_info:
+ await guardrail.make_bedrock_api_request(
+ source="INPUT",
+ messages=request_data.get("messages"),
+ request_data=request_data,
+ )
+
+ # Verify exception details
+ assert exc_info.value.status_code == 400
+ assert "Violated guardrail policy" in str(exc_info.value.detail)
+
+ print("✅ BLOCKED content with masking enabled raises exception correctly")
+
diff --git a/tests/test_litellm/test_lazy_imports.py b/tests/test_litellm/test_lazy_imports.py
index 08737df4e4a..619d9a1d1b5 100644
--- a/tests/test_litellm/test_lazy_imports.py
+++ b/tests/test_litellm/test_lazy_imports.py
@@ -14,13 +14,21 @@ from litellm._lazy_imports import (
UTILS_NAMES,
TOKEN_COUNTER_NAMES,
CACHING_NAMES,
+ BEDROCK_TYPES_NAMES,
+ TYPES_UTILS_NAMES,
+ LLM_CLIENT_CACHE_NAMES,
HTTP_HANDLER_NAMES,
_lazy_import_cost_calculator,
_lazy_import_litellm_logging,
_lazy_import_utils,
_lazy_import_token_counter,
+ _lazy_import_bedrock_types,
+ _lazy_import_types_utils,
_lazy_import_caching,
+ _lazy_import_llm_client_cache,
_lazy_import_http_handlers,
+ DOTPROMPT_NAMES,
+ _lazy_import_dotprompt,
)
@@ -111,6 +119,42 @@ def test_token_counter_lazy_imports():
_verify_only_requested_name_imported(name, TOKEN_COUNTER_NAMES)
+def test_bedrock_types_lazy_imports():
+ """Test that Bedrock type aliases can be lazy imported."""
+ for name in BEDROCK_TYPES_NAMES:
+ _clear_names_from_globals(BEDROCK_TYPES_NAMES)
+
+ alias = _lazy_import_bedrock_types(name)
+ assert alias is not None
+ assert name in litellm.__dict__
+
+ _verify_only_requested_name_imported(name, BEDROCK_TYPES_NAMES)
+
+
+def test_types_utils_lazy_imports():
+ """Test that common types.utils symbols can be lazy imported."""
+ for name in TYPES_UTILS_NAMES:
+ _clear_names_from_globals(TYPES_UTILS_NAMES)
+
+ obj = _lazy_import_types_utils(name)
+ assert obj is not None
+ assert name in litellm.__dict__
+
+ _verify_only_requested_name_imported(name, TYPES_UTILS_NAMES)
+
+
+def test_llm_client_cache_lazy_imports():
+ """Test that LLM client cache class and singleton can be lazy imported."""
+ for name in LLM_CLIENT_CACHE_NAMES:
+ _clear_names_from_globals(LLM_CLIENT_CACHE_NAMES)
+
+ obj = _lazy_import_llm_client_cache(name)
+ assert obj is not None
+ assert name in litellm.__dict__
+
+ _verify_only_requested_name_imported(name, LLM_CLIENT_CACHE_NAMES)
+
+
def test_http_handler_lazy_imports():
"""Test that HTTP handler singletons can be lazy imported."""
for name in HTTP_HANDLER_NAMES:
@@ -123,6 +167,21 @@ def test_http_handler_lazy_imports():
_verify_only_requested_name_imported(name, HTTP_HANDLER_NAMES)
+def test_dotprompt_lazy_imports():
+ """Test that dotprompt globals can be lazy imported."""
+ for name in DOTPROMPT_NAMES:
+ _clear_names_from_globals(DOTPROMPT_NAMES)
+
+ obj = _lazy_import_dotprompt(name)
+ assert name in litellm.__dict__
+
+ # Only the setter must be callable; others may be None by default
+ if name == "set_global_prompt_directory":
+ assert callable(obj), f"{name} should be callable"
+
+ _verify_only_requested_name_imported(name, DOTPROMPT_NAMES)
+
+
def test_unknown_attribute_raises_error():
"""Test that unknown attributes raise AttributeError."""
with pytest.raises(AttributeError):
@@ -140,3 +199,12 @@ def test_unknown_attribute_raises_error():
with pytest.raises(AttributeError):
_lazy_import_token_counter("unknown")
+ with pytest.raises(AttributeError):
+ _lazy_import_llm_client_cache("unknown")
+
+ with pytest.raises(AttributeError):
+ _lazy_import_bedrock_types("unknown")
+
+ with pytest.raises(AttributeError):
+ _lazy_import_types_utils("unknown")
+
diff --git a/ui/litellm-dashboard/public/assets/logos/milvus.svg b/ui/litellm-dashboard/public/assets/logos/milvus.svg
new file mode 100644
index 00000000000..76154467b4b
--- /dev/null
+++ b/ui/litellm-dashboard/public/assets/logos/milvus.svg
@@ -0,0 +1 @@
+
\ No newline at end of file
diff --git a/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.test.tsx b/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.test.tsx
new file mode 100644
index 00000000000..97d18e33640
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.test.tsx
@@ -0,0 +1,27 @@
+import { render, screen } from "@testing-library/react";
+import { describe, expect, it, vi } from "vitest";
+import { CredentialItem } from "../networking";
+import VectorStoreForm from "./VectorStoreForm";
+
+vi.mock("../networking");
+
+describe("VectorStoreForm", () => {
+ it("should render the form when visible", () => {
+ const mockOnCancel = vi.fn();
+ const mockOnSuccess = vi.fn();
+ const mockAccessToken = "test-token";
+ const mockCredentials: CredentialItem[] = [];
+
+ render(
+ ,
+ );
+
+ expect(screen.getByText("Add New Vector Store")).toBeInTheDocument();
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.tsx b/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.tsx
index d1cd3c5e443..8be879b2239 100644
--- a/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.tsx
+++ b/ui/litellm-dashboard/src/components/vector_store_management/VectorStoreForm.tsx
@@ -1,4 +1,4 @@
-import React, { useState } from "react";
+import React, { useState, useEffect } from "react";
import { TextInput, Button as TremorButton } from "@tremor/react";
import { Modal, Form, Select, Tooltip, Input, Alert } from "antd";
import { InfoCircleOutlined } from "@ant-design/icons";
@@ -10,6 +10,7 @@ import {
getProviderSpecificFields,
VectorStoreFieldConfig,
} from "../vector_store_providers";
+import { fetchAvailableModels, ModelGroup } from "../playground/llm_calls/fetch_models";
import NotificationsManager from "../molecules/notifications_manager";
interface VectorStoreFormProps {
@@ -30,6 +31,24 @@ const VectorStoreForm: React.FC = ({
const [form] = Form.useForm();
const [metadataJson, setMetadataJson] = useState("{}");
const [selectedProvider, setSelectedProvider] = useState("bedrock");
+ const [modelInfo, setModelInfo] = useState([]);
+
+ useEffect(() => {
+ if (!accessToken) return;
+
+ const loadModels = async () => {
+ try {
+ const uniqueModels = await fetchAvailableModels(accessToken);
+ if (uniqueModels.length > 0) {
+ setModelInfo(uniqueModels);
+ }
+ } catch (error) {
+ console.error("Error fetching model info:", error);
+ }
+ };
+
+ loadModels();
+ }, [accessToken]);
const handleCreate = async (formValues: any) => {
if (!accessToken) return;
@@ -207,23 +226,62 @@ const VectorStoreForm: React.FC = ({
{/* Provider-specific fields */}
- {getProviderSpecificFields(selectedProvider).map((field: VectorStoreFieldConfig) => (
-
- {field.label}{" "}
-
-
-
-
- }
- name={field.name}
- rules={field.required ? [{ required: true, message: `Please input the ${field.label.toLowerCase()}` }] : []}
- >
-
-
- ))}
+ {getProviderSpecificFields(selectedProvider).map((field: VectorStoreFieldConfig) => {
+ if (field.type === "select") {
+ const embeddingModels = modelInfo
+ .filter((option: ModelGroup) => option.mode === "embedding")
+ .map((option: ModelGroup) => ({
+ value: option.model_group,
+ label: option.model_group,
+ }));
+
+ return (
+
+ {field.label}{" "}
+
+
+
+
+ }
+ name={field.name}
+ rules={
+ field.required ? [{ required: true, message: `Please select the ${field.label.toLowerCase()}` }] : []
+ }
+ >
+
+ );
+ }
+
+ return (
+
+ {field.label}{" "}
+
+
+
+
+ }
+ name={field.name}
+ rules={
+ field.required ? [{ required: true, message: `Please input the ${field.label.toLowerCase()}` }] : []
+ }
+ >
+
+
+ );
+ })}
= {
@@ -12,6 +13,7 @@ export const vectorStoreProviderMap: Record = {
VertexRagEngine: "vertex_ai",
OpenAI: "openai",
Azure: "azure",
+ Milvus: "milvus",
};
const asset_logos_folder = "../ui/assets/logos/";
@@ -22,6 +24,7 @@ export const vectorStoreProviderLogoMap: Record = {
[VectorStoreProviders.VertexRagEngine]: `${asset_logos_folder}google.svg`,
[VectorStoreProviders.OpenAI]: `${asset_logos_folder}openai_small.svg`,
[VectorStoreProviders.Azure]: `${asset_logos_folder}microsoft_azure.svg`,
+ [VectorStoreProviders.Milvus]: `${asset_logos_folder}milvus.svg`,
};
// Define field types for provider-specific configurations
@@ -31,7 +34,7 @@ export interface VectorStoreFieldConfig {
tooltip: string;
placeholder?: string;
required: boolean;
- type?: "text" | "password";
+ type?: "text" | "password" | "select";
}
// Provider-specific field configurations
@@ -84,6 +87,33 @@ export const vectorStoreProviderFields: Record
type: "text",
},
],
+ milvus: [
+ {
+ name: "api_key",
+ label: "API Key",
+ tooltip:
+ "To obtain a token, you should use a colon (:) to concatenate the username and password that you use to access your Milvus instance (e.g., username:password)",
+ placeholder: "username:password or api key",
+ required: true,
+ type: "password",
+ },
+ {
+ name: "api_base",
+ label: "API Base",
+ tooltip: "Enter your Milvus endpoint (e.g., https://your-milvus-endpoint.com/)",
+ placeholder: "https://your-milvus-endpoint.com/",
+ required: true,
+ type: "text",
+ },
+ {
+ name: "embedding_model",
+ label: "Embedding Model",
+ tooltip: "Select the embedding model to use",
+ placeholder: "text-embedding-3-small",
+ required: true,
+ type: "select",
+ },
+ ],
};
export const getVectorStoreProviderLogoAndName = (providerValue: string): { logo: string; displayName: string } => {