mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge branch 'main' into litellm_staging_12_16_2025
This commit is contained in:
commit
215857cce3
110 changed files with 3164 additions and 343 deletions
12
.github/ISSUE_TEMPLATE/bug_report.yml
vendored
12
.github/ISSUE_TEMPLATE/bug_report.yml
vendored
|
|
@ -23,13 +23,15 @@ body:
|
|||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||
render: shell
|
||||
- type: dropdown
|
||||
id: ml-ops-team
|
||||
id: component
|
||||
attributes:
|
||||
label: Are you a ML Ops Team?
|
||||
description: This helps us prioritize your requests correctly
|
||||
label: What part of LiteLLM is this about?
|
||||
options:
|
||||
- "No"
|
||||
- "Yes"
|
||||
- "SDK (litellm Python package)"
|
||||
- "Proxy"
|
||||
- "UI Dashboard"
|
||||
- "Docs"
|
||||
- "Other"
|
||||
validations:
|
||||
required: true
|
||||
- type: input
|
||||
|
|
|
|||
12
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
12
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
|
|
@ -22,6 +22,18 @@ body:
|
|||
description: Please outline the motivation for the proposal. Is your feature request related to a specific problem? e.g., "I'm working on X and would like Y to be possible". If this is related to another GitHub issue, please link here too.
|
||||
validations:
|
||||
required: true
|
||||
- type: dropdown
|
||||
id: component
|
||||
attributes:
|
||||
label: What part of LiteLLM is this about?
|
||||
options:
|
||||
- "SDK (litellm Python package)"
|
||||
- "Proxy"
|
||||
- "UI Dashboard"
|
||||
- "Docs"
|
||||
- "Other"
|
||||
validations:
|
||||
required: true
|
||||
- type: dropdown
|
||||
id: hiring-interest
|
||||
attributes:
|
||||
|
|
|
|||
5
.github/pull_request_template.md
vendored
5
.github/pull_request_template.md
vendored
|
|
@ -1,7 +1,3 @@
|
|||
## Title
|
||||
|
||||
<!-- e.g. "Implement user authentication feature" -->
|
||||
|
||||
## Relevant issues
|
||||
|
||||
<!-- e.g. "Fixes #000" -->
|
||||
|
|
@ -11,7 +7,6 @@
|
|||
**Please complete all items before asking a LiteLLM maintainer to review your PR**
|
||||
|
||||
- [ ] I have Added testing in the [`tests/litellm/`](https://github.com/BerriAI/litellm/tree/main/tests/litellm) directory, **Adding at least 1 test is a hard requirement** - [see details](https://docs.litellm.ai/docs/extras/contributing_code)
|
||||
- [ ] I have added a screenshot of my new test passing locally
|
||||
- [ ] My PR passes all unit tests on [`make test-unit`](https://docs.litellm.ai/docs/extras/contributing_code)
|
||||
- [ ] My PR's scope is as isolated as possible, it only solves 1 specific problem
|
||||
|
||||
|
|
|
|||
43
.github/workflows/create_daily_staging_branch.yml
vendored
Normal file
43
.github/workflows/create_daily_staging_branch.yml
vendored
Normal file
|
|
@ -0,0 +1,43 @@
|
|||
name: Create Daily Staging Branch
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 0 * * *' # Runs daily at midnight UTC
|
||||
workflow_dispatch: # Allow manual trigger
|
||||
|
||||
jobs:
|
||||
create-staging-branch:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Create daily staging branch
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
# Configure Git user
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "github-actions[bot]@users.noreply.github.com"
|
||||
|
||||
# Generate branch name with MM_DD_YYYY format
|
||||
BRANCH_NAME="litellm_staging_$(date +'%m_%d_%Y')"
|
||||
echo "Creating branch: $BRANCH_NAME"
|
||||
|
||||
# Fetch all branches
|
||||
git fetch --all
|
||||
|
||||
# Check if the branch already exists
|
||||
if git show-ref --verify --quiet refs/remotes/origin/$BRANCH_NAME; then
|
||||
echo "Branch $BRANCH_NAME already exists. Skipping creation."
|
||||
else
|
||||
echo "Creating new branch: $BRANCH_NAME"
|
||||
# Create the new branch from main
|
||||
git checkout -b $BRANCH_NAME origin/main
|
||||
# Push the new branch
|
||||
git push origin $BRANCH_NAME
|
||||
echo "Successfully created and pushed branch: $BRANCH_NAME"
|
||||
fi
|
||||
2
.github/workflows/issue-keyword-labeler.yml
vendored
2
.github/workflows/issue-keyword-labeler.yml
vendored
|
|
@ -19,7 +19,7 @@ jobs:
|
|||
id: scan
|
||||
env:
|
||||
PROVIDER_ISSUE_WEBHOOK_URL: ${{ secrets.PROVIDER_ISSUE_WEBHOOK_URL }}
|
||||
KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic
|
||||
KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic,gemini,cohere,mistral,groq,ollama,deepseek
|
||||
run: python3 .github/scripts/scan_keywords.py
|
||||
|
||||
- name: Ensure label exists
|
||||
|
|
|
|||
144
.github/workflows/label-component.yml
vendored
Normal file
144
.github/workflows/label-component.yml
vendored
Normal file
|
|
@ -0,0 +1,144 @@
|
|||
name: Label Component Issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types:
|
||||
- opened
|
||||
|
||||
jobs:
|
||||
add-component-label:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
steps:
|
||||
- name: Add SDK label
|
||||
if: contains(github.event.issue.body, 'SDK (litellm Python package)')
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
script: |
|
||||
const labelName = 'sdk';
|
||||
try {
|
||||
await github.rest.issues.getLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName
|
||||
});
|
||||
} catch (error) {
|
||||
if (error.status === 404) {
|
||||
await github.rest.issues.createLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName,
|
||||
color: '0E7C86',
|
||||
description: 'Issues related to the litellm Python SDK'
|
||||
});
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.issue.number,
|
||||
labels: [labelName]
|
||||
});
|
||||
|
||||
- name: Add Proxy label
|
||||
if: contains(github.event.issue.body, 'Proxy')
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
script: |
|
||||
const labelName = 'proxy';
|
||||
try {
|
||||
await github.rest.issues.getLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName
|
||||
});
|
||||
} catch (error) {
|
||||
if (error.status === 404) {
|
||||
await github.rest.issues.createLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName,
|
||||
color: '5319E7',
|
||||
description: 'Issues related to the LiteLLM Proxy'
|
||||
});
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.issue.number,
|
||||
labels: [labelName]
|
||||
});
|
||||
|
||||
- name: Add UI Dashboard label
|
||||
if: contains(github.event.issue.body, 'UI Dashboard')
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
script: |
|
||||
const labelName = 'ui-dashboard';
|
||||
try {
|
||||
await github.rest.issues.getLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName
|
||||
});
|
||||
} catch (error) {
|
||||
if (error.status === 404) {
|
||||
await github.rest.issues.createLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName,
|
||||
color: 'D876E3',
|
||||
description: 'Issues related to the LiteLLM UI Dashboard'
|
||||
});
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.issue.number,
|
||||
labels: [labelName]
|
||||
});
|
||||
|
||||
- name: Add Docs label
|
||||
if: contains(github.event.issue.body, 'Docs')
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
script: |
|
||||
const labelName = 'docs';
|
||||
try {
|
||||
await github.rest.issues.getLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName
|
||||
});
|
||||
} catch (error) {
|
||||
if (error.status === 404) {
|
||||
await github.rest.issues.createLabel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
name: labelName,
|
||||
color: 'FBCA04',
|
||||
description: 'Issues related to LiteLLM documentation'
|
||||
});
|
||||
} else {
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.issue.number,
|
||||
labels: [labelName]
|
||||
});
|
||||
17
.github/workflows/label-mlops.yml
vendored
17
.github/workflows/label-mlops.yml
vendored
|
|
@ -1,17 +0,0 @@
|
|||
name: Label ML Ops Team Issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types:
|
||||
- opened
|
||||
|
||||
jobs:
|
||||
add-mlops-label:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check if ML Ops Team is selected
|
||||
uses: actions-ecosystem/action-add-labels@v1
|
||||
if: contains(github.event.issue.body, '### Are you a ML Ops Team?') && contains(github.event.issue.body, 'Yes')
|
||||
with:
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
labels: "mlops user request"
|
||||
|
|
@ -29,7 +29,7 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
|||
| `masterkey` | The Master API Key for LiteLLM. If not specified, a random key in the `sk-...` format is generated. | N/A |
|
||||
| `environmentSecrets` | An optional array of Secret object names. The keys and values in these secrets will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
||||
| `environmentConfigMaps` | An optional array of ConfigMap object names. The keys and values in these configmaps will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
||||
| `image.repository` | LiteLLM Proxy image repository | `ghcr.io/berriai/litellm` |
|
||||
| `image.repository` | LiteLLM Proxy image repository | `docker.litellm.ai/berriai/litellm` |
|
||||
| `image.pullPolicy` | LiteLLM Proxy image pull policy | `IfNotPresent` |
|
||||
| `image.tag` | Overrides the image tag whose default the latest version of LiteLLM at the time this chart was published. | `""` |
|
||||
| `imagePullSecrets` | Registry credentials for the LiteLLM and initContainer images. | `[]` |
|
||||
|
|
|
|||
46
docker-compose.hardened.yml
Normal file
46
docker-compose.hardened.yml
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
services:
|
||||
# Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints.
|
||||
# Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage.
|
||||
litellm:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile.non_root
|
||||
target: runtime
|
||||
args:
|
||||
PROXY_EXTRAS_SOURCE: "local"
|
||||
depends_on:
|
||||
- squid
|
||||
user: "101:101"
|
||||
group_add:
|
||||
- "2345"
|
||||
read_only: true
|
||||
cap_drop:
|
||||
- ALL
|
||||
security_opt:
|
||||
- no-new-privileges:true
|
||||
tmpfs:
|
||||
- /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777
|
||||
- /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777
|
||||
volumes:
|
||||
- ./proxy_server_config.yaml:/app/config.yaml:ro
|
||||
environment:
|
||||
LITELLM_NON_ROOT: "true"
|
||||
PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries"
|
||||
XDG_CACHE_HOME: "/app/cache"
|
||||
LITELLM_MIGRATION_DIR: "/app/migrations"
|
||||
HTTP_PROXY: "http://squid:3128"
|
||||
HTTPS_PROXY: "http://squid:3128"
|
||||
NO_PROXY: "localhost,127.0.0.1,db"
|
||||
command:
|
||||
- "--port"
|
||||
- "4000"
|
||||
- "--config"
|
||||
- "/app/config.yaml"
|
||||
squid:
|
||||
image: sameersbn/squid:3.5.27-2
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "3128:3128"
|
||||
tmpfs:
|
||||
- /var/spool/squid:rw,noexec,nosuid,nodev,size=64m
|
||||
- /var/log/squid:rw,noexec,nosuid,nodev,size=16m
|
||||
|
|
@ -4,7 +4,7 @@ services:
|
|||
context: .
|
||||
args:
|
||||
target: runtime
|
||||
image: ghcr.io/berriai/litellm:main-stable
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
#########################################
|
||||
## Uncomment these lines to start proxy with a config.yaml file ##
|
||||
# volumes:
|
||||
|
|
|
|||
|
|
@ -1,154 +1,183 @@
|
|||
# Base images
|
||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||
ARG PROXY_EXTRAS_SOURCE=published
|
||||
|
||||
# -----------------
|
||||
# Builder Stage
|
||||
# -----------------
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
|
||||
# Install build dependencies including Node.js for UI build
|
||||
USER root
|
||||
|
||||
# Install build dependencies with retry logic (includes node for UI build)
|
||||
RUN for i in 1 2 3; do \
|
||||
apk add --no-cache \
|
||||
python3 \
|
||||
py3-pip \
|
||||
clang \
|
||||
llvm \
|
||||
lld \
|
||||
gcc \
|
||||
linux-headers \
|
||||
build-base \
|
||||
bash \
|
||||
nodejs \
|
||||
npm && break || sleep 5; \
|
||||
done \
|
||||
apk add --no-cache \
|
||||
python3 \
|
||||
py3-pip \
|
||||
clang \
|
||||
llvm \
|
||||
lld \
|
||||
gcc \
|
||||
linux-headers \
|
||||
build-base \
|
||||
bash \
|
||||
nodejs \
|
||||
npm && break || sleep 5; \
|
||||
done \
|
||||
&& pip install --no-cache-dir --upgrade pip build
|
||||
|
||||
# Copy project files
|
||||
# Cache Python dependencies
|
||||
COPY requirements.txt .
|
||||
RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt \
|
||||
&& pip wheel --no-cache-dir --wheel-dir=/wheels/ "semantic_router==0.1.11" "aurelio-sdk==0.0.19" "PyJWT==2.9.0"
|
||||
|
||||
# Copy source after dependency layers
|
||||
COPY . .
|
||||
|
||||
# Set LITELLM_NON_ROOT flag for build time
|
||||
# Set non-root flag for build time consistency
|
||||
ENV LITELLM_NON_ROOT=true
|
||||
|
||||
# Build Admin UI
|
||||
RUN mkdir -p /tmp/litellm_ui
|
||||
# Build Admin UI using the upstream command order while keeping a single RUN layer
|
||||
RUN mkdir -p /tmp/litellm_ui && \
|
||||
npm install -g npm@latest && npm cache clean --force && \
|
||||
cd /app/ui/litellm-dashboard && \
|
||||
if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
|
||||
cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
|
||||
fi && \
|
||||
rm -f package-lock.json && \
|
||||
npm install --legacy-peer-deps && \
|
||||
npm run build && \
|
||||
cp -r /app/ui/litellm-dashboard/out/* /tmp/litellm_ui/ && \
|
||||
mkdir -p /tmp/litellm_assets && \
|
||||
cp /app/litellm/proxy/logo.jpg /tmp/litellm_assets/logo.jpg && \
|
||||
( cd /tmp/litellm_ui && \
|
||||
for html_file in *.html; do \
|
||||
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
|
||||
folder_name="${html_file%.html}" && \
|
||||
mkdir -p "$folder_name" && \
|
||||
mv "$html_file" "$folder_name/index.html"; \
|
||||
fi; \
|
||||
done ) && \
|
||||
cd /app/ui/litellm-dashboard && rm -rf ./out
|
||||
|
||||
RUN npm install -g npm@latest && npm cache clean --force
|
||||
|
||||
RUN cd /app/ui/litellm-dashboard && \
|
||||
if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
|
||||
cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
|
||||
fi
|
||||
|
||||
RUN cd /app/ui/litellm-dashboard && rm -f package-lock.json
|
||||
|
||||
RUN cd /app/ui/litellm-dashboard && npm install --legacy-peer-deps
|
||||
|
||||
RUN cd /app/ui/litellm-dashboard && npm run build
|
||||
|
||||
RUN cp -r /app/ui/litellm-dashboard/out/* /tmp/litellm_ui/
|
||||
RUN mkdir -p /tmp/litellm_assets && cp /app/litellm/proxy/logo.jpg /tmp/litellm_assets/logo.jpg
|
||||
|
||||
RUN cd /tmp/litellm_ui && \
|
||||
for html_file in *.html; do \
|
||||
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
|
||||
folder_name="${html_file%.html}" && \
|
||||
mkdir -p "$folder_name" && \
|
||||
mv "$html_file" "$folder_name/index.html"; \
|
||||
fi; \
|
||||
done
|
||||
|
||||
RUN cd /app/ui/litellm-dashboard && rm -rf ./out
|
||||
|
||||
# Build package and wheel dependencies
|
||||
# Build litellm wheel and place it in wheels dir (replace any PyPI wheels)
|
||||
RUN rm -rf dist/* && python -m build && \
|
||||
pip install dist/*.whl && \
|
||||
pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
||||
rm -f /wheels/litellm-*.whl && \
|
||||
cp dist/*.whl /wheels/
|
||||
|
||||
# Optionally build local litellm-proxy-extras wheel
|
||||
RUN if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
|
||||
cd /app/litellm-proxy-extras && rm -rf dist && python -m build && \
|
||||
cp dist/*.whl /wheels/; \
|
||||
fi
|
||||
|
||||
# Pre-cache Prisma binaries in the builder stage
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
|
||||
PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
|
||||
XDG_CACHE_HOME=/app/.cache \
|
||||
PATH="/usr/lib/python3.13/site-packages/nodejs/bin:${PATH}"
|
||||
|
||||
RUN pip install --no-cache-dir prisma==0.11.0 nodejs-bin==18.4.0a4 \
|
||||
&& mkdir -p /app/.cache/npm
|
||||
|
||||
RUN NPM_CONFIG_CACHE=/app/.cache/npm \
|
||||
python -c "import prisma.cli.prisma as p; p.ensure_cached()"
|
||||
|
||||
RUN prisma generate && \
|
||||
prisma --version && \
|
||||
prisma migrate diff --from-empty --to-schema-datamodel ./schema.prisma --script > /dev/null 2>&1 || true
|
||||
|
||||
# -----------------
|
||||
# Runtime Stage
|
||||
# -----------------
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
|
||||
# Install runtime dependencies
|
||||
USER root
|
||||
RUN for i in 1 2 3; do \
|
||||
apk upgrade --no-cache && break || sleep 5; \
|
||||
done \
|
||||
&& for i in 1 2 3; do \
|
||||
apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
|
||||
done
|
||||
|
||||
# Copy only necessary artifacts from builder stage for runtime
|
||||
COPY . .
|
||||
# Install runtime dependencies with retry
|
||||
RUN for i in 1 2 3; do \
|
||||
apk upgrade --no-cache && break || sleep 5; \
|
||||
done \
|
||||
&& for i in 1 2 3; do \
|
||||
apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
|
||||
done
|
||||
|
||||
# Copy artifacts from builder
|
||||
COPY --from=builder /app/requirements.txt /app/requirements.txt
|
||||
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
|
||||
COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/dist/*.whl .
|
||||
COPY --from=builder /app/schema.prisma /app/
|
||||
COPY --from=builder /wheels/ /wheels/
|
||||
COPY --from=builder /tmp/litellm_ui /tmp/litellm_ui
|
||||
COPY --from=builder /tmp/litellm_assets /tmp/litellm_assets
|
||||
COPY --from=builder /app/.cache /app/.cache
|
||||
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
|
||||
COPY --from=builder \
|
||||
/usr/lib/python3.13/site-packages/nodejs* \
|
||||
/usr/lib/python3.13/site-packages/prisma* \
|
||||
/usr/lib/python3.13/site-packages/tomlkit* \
|
||||
/usr/lib/python3.13/site-packages/nodeenv* \
|
||||
/usr/lib/python3.13/site-packages/
|
||||
COPY --from=builder /usr/bin/prisma /usr/bin/prisma
|
||||
|
||||
# Install package from wheel and dependencies
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
|
||||
&& rm -f *.whl \
|
||||
&& rm -rf /wheels
|
||||
# Final runtime environment configuration
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
|
||||
PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
|
||||
HOME=/app \
|
||||
LITELLM_NON_ROOT=true \
|
||||
XDG_CACHE_HOME=/app/.cache
|
||||
|
||||
# Remove test files and keys from dependencies
|
||||
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
||||
find /usr/lib -type d -path "*/tornado/test" -delete
|
||||
# Install packages from wheels and optional extras without network
|
||||
RUN pip install --no-index --find-links=/wheels/ -r requirements.txt && \
|
||||
pip install --no-index --find-links=/wheels/ /wheels/litellm-*-py3-none-any.whl && \
|
||||
pip install --no-index --find-links=/wheels/ --no-deps semantic_router==0.1.11 && \
|
||||
pip install --no-index --find-links=/wheels/ aurelio-sdk==0.0.19 && \
|
||||
if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
|
||||
if ls /wheels/litellm_proxy_extras-*.whl >/dev/null 2>&1; then \
|
||||
pip install --no-index --find-links=/wheels/ /wheels/litellm_proxy_extras-*.whl; \
|
||||
else \
|
||||
echo "litellm_proxy_extras wheel not found; skipping local install"; \
|
||||
fi; \
|
||||
fi
|
||||
|
||||
# Install semantic_router and aurelio-sdk using script
|
||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
||||
# Permissions, cleanup, and Prisma prep
|
||||
RUN chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh && \
|
||||
mkdir -p /nonexistent /.npm /tmp/litellm_assets /tmp/litellm_ui && \
|
||||
chown -R nobody:nogroup /app /tmp/litellm_ui /tmp/litellm_assets /nonexistent /.npm && \
|
||||
pip uninstall jwt -y || true && \
|
||||
pip uninstall PyJWT -y || true && \
|
||||
pip install --no-index --find-links=/wheels/ PyJWT==2.10.1 --no-cache-dir && \
|
||||
rm -rf /wheels && \
|
||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
chown -R nobody:nogroup $PRISMA_PATH && \
|
||||
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
||||
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH && \
|
||||
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
||||
chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
chmod -R g=u $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
chmod -R g+w $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
chmod -R g+rX $PRISMA_PATH && \
|
||||
chmod -R g+rX /app/.cache && \
|
||||
mkdir -p /tmp/.npm /nonexistent /.npm && \
|
||||
prisma generate
|
||||
|
||||
# Ensure correct JWT library is used (pyjwt not jwt)
|
||||
RUN pip uninstall jwt -y && \
|
||||
pip uninstall PyJWT -y && \
|
||||
pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# Set Prisma cache directories
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
|
||||
ENV NPM_CONFIG_CACHE=/.npm
|
||||
|
||||
# Install prisma and make entrypoints executable
|
||||
RUN pip install --no-cache-dir prisma && \
|
||||
chmod +x docker/entrypoint.sh && \
|
||||
chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
# Create directories and set permissions for non-root user
|
||||
RUN mkdir -p /nonexistent /.npm /tmp/litellm_assets && \
|
||||
chown -R nobody:nogroup /app /tmp/litellm_ui /tmp/litellm_assets /nonexistent /.npm && \
|
||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
chown -R nobody:nogroup $PRISMA_PATH && \
|
||||
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
||||
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH
|
||||
|
||||
# OpenShift compatibility
|
||||
RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
||||
chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
chmod -R g=u $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
chmod -R g+w $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true
|
||||
|
||||
# Switch to non-root user
|
||||
# Switch to non-root user for runtime
|
||||
USER nobody
|
||||
|
||||
# Set HOME for prisma generate to have a writable directory
|
||||
ENV HOME=/app
|
||||
|
||||
# Set LITELLM_NON_ROOT flag for runtime
|
||||
ENV LITELLM_NON_ROOT=true
|
||||
|
||||
RUN prisma generate
|
||||
# Prisma runtime knobs for offline containers
|
||||
ENV PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
|
||||
PRISMA_HIDE_UPDATE_MESSAGE=1 \
|
||||
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
|
||||
NPM_CONFIG_CACHE=/app/.cache/npm \
|
||||
NPM_CONFIG_PREFER_OFFLINE=true \
|
||||
PRISMA_OFFLINE_MODE=true
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||
|
||||
CMD ["--port", "4000"]
|
||||
CMD ["--port", "4000"]
|
||||
|
|
|
|||
|
|
@ -59,6 +59,30 @@ To stop the running containers, use the following command:
|
|||
docker compose down
|
||||
```
|
||||
|
||||
## Hardened / Offline Testing
|
||||
|
||||
To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file:
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache
|
||||
docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d
|
||||
```
|
||||
|
||||
This setup:
|
||||
- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image.
|
||||
- Runs the proxy as a non-root user with a read-only rootfs and only two writable tmpfs mounts:
|
||||
- `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`)
|
||||
- `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`)
|
||||
- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines.
|
||||
|
||||
You should also verify offline Prisma behaviour with:
|
||||
|
||||
```bash
|
||||
docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version
|
||||
```
|
||||
|
||||
This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ Add A2A Agents on LiteLLM AI Gateway, Invoke agents in A2A Protocol, track reque
|
|||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Supported Agent Providers | A2A, LangGraph, Azure AI Foundry, Bedrock AgentCore |
|
||||
| Supported Agent Providers | A2A, Vertex AI Agent Engine, LangGraph, Azure AI Foundry, Bedrock AgentCore, Pydantic AI |
|
||||
| Logging | ✅ |
|
||||
| Load Balancing | ✅ |
|
||||
| Streaming | ✅ |
|
||||
|
|
@ -45,17 +45,26 @@ You can add A2A-compatible agents through the LiteLLM Admin UI.
|
|||
|
||||
The URL should be the invocation URL for your A2A agent (e.g., `http://localhost:10001`).
|
||||
|
||||
|
||||
### Add Azure AI Foundry Agents
|
||||
|
||||
Follow [this guide, to add your azure ai foundry agent to LiteLLM Agent Gateway](./providers/azure_ai_agents#litellm-a2a-gateway)
|
||||
|
||||
### Add Vertex AI Agent Engine
|
||||
|
||||
Follow [this guide, to add your Vertex AI Agent Engine to LiteLLM Agent Gateway](./providers/vertex_ai_agent_engine)
|
||||
|
||||
### Add Bedrock AgentCore Agents
|
||||
|
||||
Follow [this guide, to add your bedrock agentcore agent to LiteLLM Agent Gateway](./providers/bedrock_agentcore#litellm-a2a-gateway)
|
||||
|
||||
### Add LangGraph Agents
|
||||
|
||||
Follow [this guide, to add your langgraph agent to LiteLLM Agent Gateway](./providers/langgraph#litellm-a2a-gateway)
|
||||
|
||||
### Add Bedrock AgentCore Agents
|
||||
### Add Pydantic AI Agents
|
||||
|
||||
Follow [this guide, to add your bedrock agentcore agent to LiteLLM Agent Gateway](./providers/bedrock_agentcore#litellm-a2a-gateway)
|
||||
Follow [this guide, to add your pydantic ai agent to LiteLLM Agent Gateway](./providers/pydantic_ai_agent#litellm-a2a-gateway)
|
||||
|
||||
## Invoking your Agents
|
||||
|
||||
|
|
|
|||
|
|
@ -657,7 +657,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -181,7 +181,7 @@ docker run \
|
|||
-e USE_DDTRACE=true \
|
||||
-e USE_DDPROFILER=true \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
|
|||
121
docs/my-website/docs/providers/pydantic_ai_agent.md
Normal file
121
docs/my-website/docs/providers/pydantic_ai_agent.md
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Pydantic AI Agents
|
||||
|
||||
Call Pydantic AI Agents via LiteLLM's A2A Gateway.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Pydantic AI agents with native A2A support via the `to_a2a()` method. LiteLLM provides fake streaming support for agents that don't natively stream. |
|
||||
| Provider Route on LiteLLM | A2A Gateway |
|
||||
| Supported Endpoints | `/v1/a2a/message/send` |
|
||||
| Provider Doc | [Pydantic AI Agents ↗](https://ai.pydantic.dev/agents/) |
|
||||
|
||||
## LiteLLM A2A Gateway
|
||||
|
||||
All Pydantic AI agents need to be exposed as A2A agents using the `to_a2a()` method. Once your agent server is running, you can add it to the LiteLLM Gateway.
|
||||
|
||||
### 1. Setup Pydantic AI Agent Server
|
||||
|
||||
LiteLLM requires Pydantic AI agents to follow the [A2A (Agent-to-Agent) protocol](https://github.com/google/A2A). Pydantic AI has native A2A support via the `to_a2a()` method, which exposes your agent as an A2A-compliant server.
|
||||
|
||||
#### Install Dependencies
|
||||
|
||||
```bash
|
||||
pip install pydantic-ai fasta2a uvicorn
|
||||
```
|
||||
|
||||
#### Create Agent
|
||||
|
||||
```python title="agent.py"
|
||||
from pydantic_ai import Agent
|
||||
|
||||
agent = Agent('openai:gpt-4o-mini', instructions='Be helpful!')
|
||||
|
||||
@agent.tool_plain
|
||||
def get_weather(city: str) -> str:
|
||||
"""Get weather for a city."""
|
||||
return f"Weather in {city}: Sunny, 72°F"
|
||||
|
||||
@agent.tool_plain
|
||||
def calculator(expression: str) -> str:
|
||||
"""Evaluate a math expression."""
|
||||
return str(eval(expression))
|
||||
|
||||
# Native A2A server - Pydantic AI handles it automatically
|
||||
app = agent.to_a2a()
|
||||
```
|
||||
|
||||
#### Run Server
|
||||
|
||||
```bash
|
||||
uvicorn agent:app --host 0.0.0.0 --port 9999
|
||||
```
|
||||
|
||||
Server runs at `http://localhost:9999`
|
||||
|
||||
### 2. Navigate to Agents
|
||||
|
||||
From the sidebar, click "Agents" to open the agent management page, then click "+ Add New Agent".
|
||||
|
||||
### 3. Select Pydantic AI Agent Type
|
||||
|
||||
Click "A2A Standard" to see available agent types, then select "Pydantic AI".
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 4. Configure the Agent
|
||||
|
||||
Fill in the following fields:
|
||||
|
||||
- **Agent Name** - A unique identifier for your agent (e.g., `test-pydantic-agent`)
|
||||
- **Agent URL** - The URL where your Pydantic AI agent is running. We use `http://localhost:9999` because that's where we started our Pydantic AI agent server in the previous step.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 5. Create Agent
|
||||
|
||||
Click "Create Agent" to save your configuration.
|
||||
|
||||

|
||||
|
||||
### 6. Test in Playground
|
||||
|
||||
Go to "Playground" in the sidebar to test your agent.
|
||||
|
||||

|
||||
|
||||
### 7. Select A2A Endpoint
|
||||
|
||||
Click the endpoint dropdown and search for "a2a", then select `/v1/a2a/message/send`.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 8. Select Your Agent and Send a Message
|
||||
|
||||
Pick your Pydantic AI agent from the dropdown and send a test message.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Pydantic AI Documentation](https://ai.pydantic.dev/)
|
||||
- [Pydantic AI Agents](https://ai.pydantic.dev/agents/)
|
||||
- [A2A Agent Gateway](../a2a.md)
|
||||
- [A2A Cost Tracking](../a2a_cost_tracking.md)
|
||||
216
docs/my-website/docs/providers/vertex_ai_agent_engine.md
Normal file
216
docs/my-website/docs/providers/vertex_ai_agent_engine.md
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Vertex AI Agent Engine
|
||||
|
||||
Call Vertex AI Agent Engine (Reasoning Engines) in the OpenAI Request/Response format.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Vertex AI Agent Engine provides hosted agent runtimes that can execute agentic workflows with foundation models, tools, and custom logic. |
|
||||
| Provider Route on LiteLLM | `vertex_ai/agent_engine/{RESOURCE_NAME}` |
|
||||
| Supported Endpoints | `/chat/completions`, `/v1/messages`, `/v1/responses`, `/v1/a2a/message/send` |
|
||||
| Provider Doc | [Vertex AI Agent Engine ↗](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/overview) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Model Format
|
||||
|
||||
```shell showLineNumbers title="Model Format"
|
||||
vertex_ai/agent_engine/{RESOURCE_NAME}
|
||||
```
|
||||
|
||||
**Example:**
|
||||
- `vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888`
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="Basic Agent Completion"
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888",
|
||||
messages=[
|
||||
{"role": "user", "content": "Explain machine learning in simple terms"}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Streaming Agent Responses"
|
||||
import litellm
|
||||
|
||||
response = await litellm.acompletion(
|
||||
model="vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888",
|
||||
messages=[
|
||||
{"role": "user", "content": "What are the key principles of software architecture?"}
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
async for chunk in response:
|
||||
if chunk.choices[0].delta.content:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your model in config.yaml
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: vertex-agent-1
|
||||
litellm_params:
|
||||
model: vertex_ai/agent_engine/projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Start the LiteLLM Proxy
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
#### 3. Make requests to your Vertex AI Agent Engine
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Basic Agent Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "vertex-agent-1",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Summarize the main benefits of cloud computing"}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="vertex-agent-1",
|
||||
messages=[
|
||||
{"role": "user", "content": "What are best practices for API design?"}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## LiteLLM A2A Gateway
|
||||
|
||||
You can also connect to Vertex AI Agent Engine through LiteLLM's A2A (Agent-to-Agent) Gateway UI. This provides a visual way to register and test agents without writing code.
|
||||
|
||||
### 1. Navigate to Agents
|
||||
|
||||
From the sidebar, click "Agents" to open the agent management page, then click "+ Add New Agent".
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 2. Select Vertex AI Agent Engine Type
|
||||
|
||||
Click "A2A Standard" to see available agent types, then select "Vertex AI Agent Engine".
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 3. Configure the Agent
|
||||
|
||||
Fill in the following fields:
|
||||
|
||||
- **Agent Name** - A friendly name for your agent (e.g., `my-vertex-agent`)
|
||||
- **Reasoning Engine Resource ID** - The full resource path from Google Cloud Console (e.g., `projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888`)
|
||||
- **Vertex Project** - Your Google Cloud project ID
|
||||
- **Vertex Location** - The region where your agent is deployed (e.g., `us-central1`)
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
You can find the Resource ID in Google Cloud Console under Vertex AI > Agent Engine:
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
You can find the Project ID in Google Cloud Console:
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
### 4. Create Agent
|
||||
|
||||
Click "Create Agent" to save your configuration.
|
||||
|
||||

|
||||
|
||||
### 5. Test in Playground
|
||||
|
||||
Go to "Playground" in the sidebar to test your agent.
|
||||
|
||||

|
||||
|
||||
### 6. Select A2A Endpoint
|
||||
|
||||
Click the endpoint dropdown and select `/v1/a2a/message/send`.
|
||||
|
||||

|
||||
|
||||
### 7. Select Your Agent and Send a Message
|
||||
|
||||
Pick your Vertex AI Agent Engine from the dropdown and send a test message.
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `GOOGLE_APPLICATION_CREDENTIALS` | Path to service account JSON key file |
|
||||
| `VERTEXAI_PROJECT` | Google Cloud project ID |
|
||||
| `VERTEXAI_LOCATION` | Google Cloud region (default: `us-central1`) |
|
||||
|
||||
```bash
|
||||
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service-account.json"
|
||||
export VERTEXAI_PROJECT="your-project-id"
|
||||
export VERTEXAI_LOCATION="us-central1"
|
||||
```
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [Vertex AI Agent Engine Documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/overview)
|
||||
- [Create a Reasoning Engine](https://cloud.google.com/vertex-ai/generative-ai/docs/reasoning-engine/create)
|
||||
- [A2A Agent Gateway](../a2a.md)
|
||||
- [Vertex AI Provider](./vertex.md)
|
||||
|
|
@ -655,7 +655,7 @@ docker run --name litellm-proxy \
|
|||
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="<object_key>> \
|
||||
-e LITELLM_CONFIG_BUCKET_TYPE="gcs" \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-latest --detailed_debug
|
||||
docker.litellm.ai/berriai/litellm-database:main-latest --detailed_debug
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -676,7 +676,7 @@ docker run --name litellm-proxy \
|
|||
-e LITELLM_CONFIG_BUCKET_NAME=<bucket_name> \
|
||||
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="<object_key>> \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-latest
|
||||
docker.litellm.ai/berriai/litellm-database:main-latest
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
|
|||
|
|
@ -10,10 +10,38 @@ You can find the Dockerfile to build litellm proxy [here](https://github.com/Ber
|
|||
|
||||
## Quick Start
|
||||
|
||||
:::info
|
||||
Facing issues with pulling the docker image? Email us at support@berri.ai.
|
||||
:::
|
||||
|
||||
To start using Litellm, run the following commands in a shell:
|
||||
|
||||
<Tabs>
|
||||
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
```
|
||||
docker pull docker.litellm.ai/berriai/litellm:main-latest
|
||||
```
|
||||
|
||||
[**See all docker images**](https://github.com/orgs/BerriAI/packages)
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="LiteLLM CLI (pip package)">
|
||||
|
||||
```shell
|
||||
$ pip install 'litellm[proxy]'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="docker-compose" label="Docker Compose (Proxy + DB)">
|
||||
|
||||
Use this docker compose to spin up the proxy with a postgres database running locally.
|
||||
|
||||
```bash
|
||||
# Get the code
|
||||
# Get the docker compose file
|
||||
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/docker-compose.yml
|
||||
curl -O https://raw.githubusercontent.com/BerriAI/litellm/main/prometheus.yml
|
||||
|
||||
|
|
@ -30,6 +58,8 @@ echo 'LITELLM_SALT_KEY="sk-1234"' >> .env
|
|||
docker compose up
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Docker Run
|
||||
|
||||
|
|
@ -57,7 +87,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-stable \
|
||||
docker.litellm.ai/berriai/litellm:main-stable \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
@ -87,12 +117,12 @@ See all supported CLI args [here](https://docs.litellm.ai/docs/proxy/cli):
|
|||
|
||||
Here's how you can run the docker image and pass your config to `litellm`
|
||||
```shell
|
||||
docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable --config your_config.yaml
|
||||
```
|
||||
|
||||
Here's how you can run the docker image and start litellm on port 8002 with `num_workers=8`
|
||||
```shell
|
||||
docker run ghcr.io/berriai/litellm:main-stable --port 8002 --num_workers 8
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable --port 8002 --num_workers 8
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -100,7 +130,7 @@ docker run ghcr.io/berriai/litellm:main-stable --port 8002 --num_workers 8
|
|||
|
||||
```shell
|
||||
# Use the provided base image
|
||||
FROM ghcr.io/berriai/litellm:main-stable
|
||||
FROM docker.litellm.ai/berriai/litellm:main-stable
|
||||
|
||||
# Set the working directory to /app
|
||||
WORKDIR /app
|
||||
|
|
@ -242,7 +272,7 @@ spec:
|
|||
spec:
|
||||
containers:
|
||||
- name: litellm
|
||||
image: ghcr.io/berriai/litellm:main-stable # it is recommended to fix a version generally
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable # it is recommended to fix a version generally
|
||||
args:
|
||||
- "--config"
|
||||
- "/app/proxy_server_config.yaml"
|
||||
|
|
@ -279,9 +309,9 @@ Use this when you want to use litellm helm chart as a dependency for other chart
|
|||
#### Step 1. Pull the litellm helm chart
|
||||
|
||||
```bash
|
||||
helm pull oci://ghcr.io/berriai/litellm-helm
|
||||
helm pull oci://docker.litellm.ai/berriai/litellm-helm
|
||||
|
||||
# Pulled: ghcr.io/berriai/litellm-helm:0.1.2
|
||||
# Pulled: docker.litellm.ai/berriai/litellm-helm:0.1.2
|
||||
# Digest: sha256:7d3ded1c99c1597f9ad4dc49d84327cf1db6e0faa0eeea0c614be5526ae94e2a
|
||||
```
|
||||
|
||||
|
|
@ -340,7 +370,7 @@ Requirements:
|
|||
We maintain a [separate Dockerfile](https://github.com/BerriAI/litellm/pkgs/container/litellm-database) for reducing build time when running LiteLLM proxy with a connected Postgres Database
|
||||
|
||||
```shell
|
||||
docker pull ghcr.io/berriai/litellm-database:main-stable
|
||||
docker pull docker.litellm.ai/berriai/litellm-database:main-stable
|
||||
```
|
||||
|
||||
```shell
|
||||
|
|
@ -351,7 +381,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-stable \
|
||||
docker.litellm.ai/berriai/litellm-database:main-stable \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
@ -379,7 +409,7 @@ spec:
|
|||
spec:
|
||||
containers:
|
||||
- name: litellm-container
|
||||
image: ghcr.io/berriai/litellm:main-stable
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
imagePullPolicy: Always
|
||||
env:
|
||||
- name: AZURE_API_KEY
|
||||
|
|
@ -516,9 +546,9 @@ Use this when you want to use litellm helm chart as a dependency for other chart
|
|||
#### Step 1. Pull the litellm helm chart
|
||||
|
||||
```bash
|
||||
helm pull oci://ghcr.io/berriai/litellm-helm
|
||||
helm pull oci://docker.litellm.ai/berriai/litellm-helm
|
||||
|
||||
# Pulled: ghcr.io/berriai/litellm-helm:0.1.2
|
||||
# Pulled: docker.litellm.ai/berriai/litellm-helm:0.1.2
|
||||
# Digest: sha256:7d3ded1c99c1597f9ad4dc49d84327cf1db6e0faa0eeea0c614be5526ae94e2a
|
||||
```
|
||||
|
||||
|
|
@ -575,7 +605,7 @@ router_settings:
|
|||
Start docker container with config
|
||||
|
||||
```shell
|
||||
docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable --config your_config.yaml
|
||||
```
|
||||
|
||||
### Deploy with Database + Redis
|
||||
|
|
@ -610,7 +640,7 @@ Start `litellm-database`docker container with config
|
|||
docker run --name litellm-proxy \
|
||||
-e DATABASE_URL=postgresql://<user>:<password>@<host>:<port>/<dbname> \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-stable --config your_config.yaml
|
||||
docker.litellm.ai/berriai/litellm-database:main-stable --config your_config.yaml
|
||||
```
|
||||
|
||||
### (Non Root) - without Internet Connection
|
||||
|
|
@ -620,7 +650,7 @@ By default `prisma generate` downloads [prisma's engine binaries](https://www.pr
|
|||
Use this docker image to deploy litellm with pre-generated prisma binaries.
|
||||
|
||||
```bash
|
||||
docker pull ghcr.io/berriai/litellm-non_root:main-stable
|
||||
docker pull docker.litellm.ai/berriai/litellm-non_root:main-stable
|
||||
```
|
||||
|
||||
[Published Docker Image link](https://github.com/BerriAI/litellm/pkgs/container/litellm-non_root)
|
||||
|
|
@ -639,7 +669,7 @@ Use this, If you need to set ssl certificates for your on prem litellm proxy
|
|||
Pass `ssl_keyfile_path` (Path to the SSL keyfile) and `ssl_certfile_path` (Path to the SSL certfile) when starting litellm proxy
|
||||
|
||||
```shell
|
||||
docker run ghcr.io/berriai/litellm:main-stable \
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable \
|
||||
--ssl_keyfile_path ssl_test/keyfile.key \
|
||||
--ssl_certfile_path ssl_test/certfile.crt
|
||||
```
|
||||
|
|
@ -654,7 +684,7 @@ Step 1. Build your custom docker image with hypercorn
|
|||
|
||||
```shell
|
||||
# Use the provided base image
|
||||
FROM ghcr.io/berriai/litellm:main-stable
|
||||
FROM docker.litellm.ai/berriai/litellm:main-stable
|
||||
|
||||
# Set the working directory to /app
|
||||
WORKDIR /app
|
||||
|
|
@ -702,7 +732,7 @@ Usage Example:
|
|||
In this example, we set the keepalive timeout to 75 seconds.
|
||||
|
||||
```shell showLineNumbers title="docker run"
|
||||
docker run ghcr.io/berriai/litellm:main-stable \
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable \
|
||||
--keepalive_timeout 75
|
||||
```
|
||||
|
||||
|
|
@ -711,7 +741,7 @@ In this example, we set the keepalive timeout to 75 seconds.
|
|||
|
||||
```shell showLineNumbers title="Environment Variable"
|
||||
export KEEPALIVE_TIMEOUT=75
|
||||
docker run ghcr.io/berriai/litellm:main-stable
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -722,7 +752,7 @@ Use this to mitigate memory growth by recycling workers after a fixed number of
|
|||
Usage Examples:
|
||||
|
||||
```shell showLineNumbers title="docker run (CLI flag)"
|
||||
docker run ghcr.io/berriai/litellm:main-stable \
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable \
|
||||
--max_requests_before_restart 10000
|
||||
```
|
||||
|
||||
|
|
@ -730,7 +760,7 @@ Or set via environment variable:
|
|||
|
||||
```shell showLineNumbers title="Environment Variable"
|
||||
export MAX_REQUESTS_BEFORE_RESTART=10000
|
||||
docker run ghcr.io/berriai/litellm:main-stable
|
||||
docker run docker.litellm.ai/berriai/litellm:main-stable
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -759,7 +789,7 @@ docker run --name litellm-proxy \
|
|||
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="<object_key>> \
|
||||
-e LITELLM_CONFIG_BUCKET_TYPE="gcs" \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-stable --detailed_debug
|
||||
docker.litellm.ai/berriai/litellm-database:main-stable --detailed_debug
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -780,7 +810,7 @@ docker run --name litellm-proxy \
|
|||
-e LITELLM_CONFIG_BUCKET_NAME=<bucket_name> \
|
||||
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="<object_key>> \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-stable
|
||||
docker.litellm.ai/berriai/litellm-database:main-stable
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
@ -907,7 +937,7 @@ Run the following command, replacing `<database_url>` with the value you copied
|
|||
docker run --name litellm-proxy \
|
||||
-e DATABASE_URL=<database_url> \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm-database:main-stable
|
||||
docker.litellm.ai/berriai/litellm-database:main-stable
|
||||
```
|
||||
|
||||
#### 4. Access the Application:
|
||||
|
|
@ -986,7 +1016,7 @@ services:
|
|||
context: .
|
||||
args:
|
||||
target: runtime
|
||||
image: ghcr.io/berriai/litellm:main-stable
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
ports:
|
||||
- "4000:4000" # Map the container port to the host, change the host port if necessary
|
||||
volumes:
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ End-to-End tutorial for LiteLLM Proxy to:
|
|||
<TabItem value="docker" label="Docker">
|
||||
|
||||
```
|
||||
docker pull ghcr.io/berriai/litellm:main-latest
|
||||
docker pull docker.litellm.ai/berriai/litellm:main-latest
|
||||
```
|
||||
|
||||
[**See all docker images**](https://github.com/orgs/BerriAI/packages)
|
||||
|
|
@ -119,7 +119,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
|
|
@ -302,7 +302,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -67,7 +67,7 @@ docker run --rm \
|
|||
-e PANGEA_AI_GUARD_TOKEN=$PANGEA_AI_GUARD_TOKEN \
|
||||
-e OPENAI_API_KEY=$OPENAI_API_KEY \
|
||||
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -29,6 +29,10 @@ LiteLLM automatically distributes requests across multiple deployments of the sa
|
|||
| **latency-based-routing** | Routes to fastest responding deployment | Latency-critical applications |
|
||||
| **cost-based-routing** | Routes to deployment with lowest cost | Cost-sensitive applications |
|
||||
|
||||
:::tip Deployment Priority
|
||||
Use the `order` parameter to prioritize specific deployments. [See Deployment Ordering](#deployment-ordering-priority) for details.
|
||||
:::
|
||||
|
||||
|
||||
## Quick Start - Load Balancing
|
||||
#### Step 1 - Set deployments on config
|
||||
|
|
@ -243,6 +247,27 @@ class RouterModelGroupAliasItem(TypedDict):
|
|||
hidden: bool # if 'True', don't return on `/v1/models`, `/v1/model/info`, `/v1/model_group/info`
|
||||
```
|
||||
|
||||
## Deployment Ordering (Priority)
|
||||
|
||||
Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them.
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: azure/gpt-4-primary
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
order: 1 # 👈 Highest priority - always tried first
|
||||
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: azure/gpt-4-fallback
|
||||
api_key: os.environ/AZURE_API_KEY_2
|
||||
order: 2 # 👈 Used when order=1 is unavailable
|
||||
```
|
||||
|
||||
If `order=1` deployment is unavailable (e.g., rate-limited), the router falls back to `order=2` deployments.
|
||||
|
||||
### When You'll See Load Balancing in Action
|
||||
|
||||
**Immediate Effects:**
|
||||
|
|
|
|||
|
|
@ -269,7 +269,7 @@ spec:
|
|||
spec:
|
||||
containers:
|
||||
- name: litellm-proxy
|
||||
image: ghcr.io/berriai/litellm:latest
|
||||
image: docker.litellm.ai/berriai/litellm:latest
|
||||
env:
|
||||
- name: USE_SHARED_HEALTH_CHECK
|
||||
value: "true"
|
||||
|
|
|
|||
|
|
@ -832,6 +832,59 @@ asyncio.run(router_acompletion())
|
|||
|
||||
## Basic Reliability
|
||||
|
||||
### Deployment Ordering (Priority)
|
||||
|
||||
Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import Router
|
||||
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-4-primary",
|
||||
"api_key": os.getenv("AZURE_API_KEY"),
|
||||
"order": 1, # 👈 Highest priority
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-4-fallback",
|
||||
"api_key": os.getenv("AZURE_API_KEY_2"),
|
||||
"order": 2, # 👈 Used when order=1 is unavailable
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
router = Router(model_list=model_list)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: azure/gpt-4-primary
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
order: 1 # 👈 Highest priority
|
||||
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: azure/gpt-4-fallback
|
||||
api_key: os.environ/AZURE_API_KEY_2
|
||||
order: 2 # 👈 Used when order=1 is unavailable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Weighted Deployments
|
||||
|
||||
Set `weight` on a deployment to pick one deployment more often than others.
|
||||
|
|
|
|||
|
|
@ -76,7 +76,7 @@ docker run -d \
|
|||
--name litellm-proxy \
|
||||
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||
-v $(pwd)/my_secret_manager.py:/app/my_secret_manager.py \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug
|
||||
|
|
|
|||
|
|
@ -221,7 +221,7 @@ services:
|
|||
- elasticsearch
|
||||
|
||||
litellm:
|
||||
image: ghcr.io/berriai/litellm:main-latest
|
||||
image: docker.litellm.ai/berriai/litellm:main-latest
|
||||
ports:
|
||||
- "4000:4000"
|
||||
environment:
|
||||
|
|
|
|||
|
|
@ -53,7 +53,7 @@ yarn global add @openai/codex
|
|||
docker run \
|
||||
-v $(pwd)/litellm_config.yaml:/app/config.yaml \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -53,7 +53,7 @@ Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable
|
||||
docker.litellm.ai/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable
|
||||
```
|
||||
|
||||
## Get Daily Updates
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-
|
|||
**You are only impacted if you use `apt-get` in your Dockerfile**
|
||||
```shell
|
||||
# Use the provided base image
|
||||
FROM ghcr.io/berriai/litellm:main-latest
|
||||
FROM docker.litellm.ai/berriai/litellm:main-latest
|
||||
|
||||
# Set the working directory
|
||||
WORKDIR /app
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ This release is primarily focused on:
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.63.11-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.63.11-stable
|
||||
```
|
||||
|
||||
## Demo Instance
|
||||
|
|
|
|||
|
|
@ -32,7 +32,7 @@ This release brings:
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1
|
||||
docker.litellm.ai/berriai/litellm:main-v1.63.14-stable.patch1
|
||||
```
|
||||
|
||||
## Demo Instance
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.65.4-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.65.4-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.66.0-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.66.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.67.4-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.67.4-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.68.0-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.68.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.69.0-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.69.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.70.1-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.70.1-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.71.1-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.71.1-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.72.0-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.72.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.72.2-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.72.2-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.72.6-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.72.6-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ The `non-root` docker image has a known issue around the UI not loading. If you
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.73.0-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.73.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.73.6-stable.patch.1
|
||||
docker.litellm.ai/berriai/litellm:v1.73.6-stable.patch.1
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.74.0-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.74.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.74.15-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.74.15-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.74.3-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.74.3-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.74.7-stable.patch.1
|
||||
docker.litellm.ai/berriai/litellm:v1.74.7-stable.patch.1
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.74.9-stable.patch.1
|
||||
docker.litellm.ai/berriai/litellm:v1.74.9-stable.patch.1
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.75.5-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.75.5-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.75.8-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.75.8-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.76.1
|
||||
docker.litellm.ai/berriai/litellm:v1.76.1
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ This release has a known issue where startup is leading to Out of Memory errors
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.76.3
|
||||
docker.litellm.ai/berriai/litellm:v1.76.3
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-v1.77.2-stable
|
||||
docker.litellm.ai/berriai/litellm:main-v1.77.2-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.77.3-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.77.3-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.77.5-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.77.5-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.77.7.rc.1
|
||||
docker.litellm.ai/berriai/litellm:v1.77.7.rc.1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.78.0-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.78.0-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.78.5-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.78.5-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.79.0-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.79.0-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.79.1-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.79.1-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.79.3-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.79.3-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.80.0-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.80.0-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.80.10.rc.1
|
||||
docker.litellm.ai/berriai/litellm:v1.80.10.rc.1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.80.5-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.80.5-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ import TabItem from '@theme/TabItem';
|
|||
docker run \
|
||||
-e STORE_MODEL_IN_DB=True \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:v1.80.8-stable
|
||||
docker.litellm.ai/berriai/litellm:v1.80.8-stable
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
|
|
@ -632,6 +632,7 @@ const sidebars = {
|
|||
"providers/vertex_speech",
|
||||
"providers/vertex_batch",
|
||||
"providers/vertex_ocr",
|
||||
"providers/vertex_ai_agent_engine",
|
||||
]
|
||||
},
|
||||
{
|
||||
|
|
@ -738,6 +739,7 @@ const sidebars = {
|
|||
"providers/petals",
|
||||
"providers/publicai",
|
||||
"providers/predibase",
|
||||
"providers/pydantic_ai_agent",
|
||||
"providers/ragflow",
|
||||
"providers/recraft",
|
||||
"providers/replicate",
|
||||
|
|
|
|||
|
|
@ -604,7 +604,7 @@ docker run \
|
|||
-e AZURE_API_KEY=d6*********** \
|
||||
-e AZURE_API_BASE=https://openai-***********/ \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
docker.litellm.ai/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -18,6 +18,45 @@ def str_to_bool(value: Optional[str]) -> bool:
|
|||
return value.lower() in ("true", "1", "t", "y", "yes")
|
||||
|
||||
|
||||
|
||||
def _get_prisma_env() -> dict:
|
||||
"""Get environment variables for Prisma, handling offline mode if configured."""
|
||||
prisma_env = os.environ.copy()
|
||||
if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
|
||||
# These env vars prevent Prisma from attempting downloads
|
||||
prisma_env["NPM_CONFIG_PREFER_OFFLINE"] = "true"
|
||||
prisma_env["NPM_CONFIG_CACHE"] = os.getenv("NPM_CONFIG_CACHE", "/app/.cache/npm")
|
||||
return prisma_env
|
||||
|
||||
|
||||
def _get_prisma_command() -> str:
|
||||
"""Get the Prisma command to use, bypassing Python wrapper in offline mode."""
|
||||
if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
|
||||
# Primary location where Prisma Python package installs the CLI
|
||||
default_cli_path = "/app/.cache/prisma-python/binaries/node_modules/.bin/prisma"
|
||||
|
||||
# Check if custom path is provided (for flexibility)
|
||||
custom_cli_path = os.getenv("PRISMA_CLI_PATH")
|
||||
if custom_cli_path and os.path.exists(custom_cli_path):
|
||||
logger.info(f"Using custom Prisma CLI at {custom_cli_path}")
|
||||
return custom_cli_path
|
||||
|
||||
# Check the default location
|
||||
if os.path.exists(default_cli_path):
|
||||
logger.info(f"Using cached Prisma CLI at {default_cli_path}")
|
||||
return default_cli_path
|
||||
|
||||
# If not found, log warning and fall back
|
||||
logger.warning(
|
||||
f"Prisma CLI not found at {default_cli_path}. "
|
||||
"Falling back to Python wrapper (may attempt downloads)"
|
||||
)
|
||||
|
||||
# Fall back to the Python wrapper (will work in online mode)
|
||||
return "prisma"
|
||||
|
||||
|
||||
|
||||
class ProxyExtrasDBManager:
|
||||
@staticmethod
|
||||
def _get_prisma_dir() -> str:
|
||||
|
|
@ -57,6 +96,11 @@ class ProxyExtrasDBManager:
|
|||
init_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
database_url = os.getenv("DATABASE_URL")
|
||||
if not database_url:
|
||||
logger.error("DATABASE_URL not set")
|
||||
return False
|
||||
# Set up environment for offline mode if configured
|
||||
prisma_env = _get_prisma_env()
|
||||
|
||||
try:
|
||||
# 1. Generate migration SQL file by comparing empty state to current db state
|
||||
|
|
@ -64,7 +108,7 @@ class ProxyExtrasDBManager:
|
|||
migration_file = init_dir / "migration.sql"
|
||||
subprocess.run(
|
||||
[
|
||||
"prisma",
|
||||
_get_prisma_command(),
|
||||
"migrate",
|
||||
"diff",
|
||||
"--from-empty",
|
||||
|
|
@ -75,13 +119,14 @@ class ProxyExtrasDBManager:
|
|||
stdout=open(migration_file, "w"),
|
||||
check=True,
|
||||
timeout=30,
|
||||
env=prisma_env
|
||||
)
|
||||
|
||||
# 3. Mark the migration as applied since it represents current state
|
||||
logger.info("Marking baseline migration as applied...")
|
||||
subprocess.run(
|
||||
[
|
||||
"prisma",
|
||||
_get_prisma_command(),
|
||||
"migrate",
|
||||
"resolve",
|
||||
"--applied",
|
||||
|
|
@ -89,6 +134,7 @@ class ProxyExtrasDBManager:
|
|||
],
|
||||
check=True,
|
||||
timeout=30,
|
||||
env=prisma_env
|
||||
)
|
||||
|
||||
return True
|
||||
|
|
@ -113,21 +159,26 @@ class ProxyExtrasDBManager:
|
|||
@staticmethod
|
||||
def _roll_back_migration(migration_name: str):
|
||||
"""Mark a specific migration as rolled back"""
|
||||
# Set up environment for offline mode if configured
|
||||
prisma_env = _get_prisma_env()
|
||||
subprocess.run(
|
||||
["prisma", "migrate", "resolve", "--rolled-back", migration_name],
|
||||
[_get_prisma_command(), "migrate", "resolve", "--rolled-back", migration_name],
|
||||
timeout=60,
|
||||
check=True,
|
||||
capture_output=True,
|
||||
env=prisma_env
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _resolve_specific_migration(migration_name: str):
|
||||
"""Mark a specific migration as applied"""
|
||||
prisma_env = _get_prisma_env()
|
||||
subprocess.run(
|
||||
["prisma", "migrate", "resolve", "--applied", migration_name],
|
||||
[_get_prisma_command(), "migrate", "resolve", "--applied", migration_name],
|
||||
timeout=60,
|
||||
check=True,
|
||||
capture_output=True,
|
||||
env=prisma_env
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -194,6 +245,10 @@ class ProxyExtrasDBManager:
|
|||
3. Mark all existing migrations as applied.
|
||||
"""
|
||||
database_url = os.getenv("DATABASE_URL")
|
||||
if not database_url:
|
||||
logger.error("DATABASE_URL not set")
|
||||
return
|
||||
|
||||
diff_dir = (
|
||||
Path(migrations_dir)
|
||||
/ "migrations"
|
||||
|
|
@ -216,7 +271,7 @@ class ProxyExtrasDBManager:
|
|||
with open(diff_sql_path, "w") as f:
|
||||
subprocess.run(
|
||||
[
|
||||
"prisma",
|
||||
_get_prisma_command(),
|
||||
"migrate",
|
||||
"diff",
|
||||
"--from-url",
|
||||
|
|
@ -228,6 +283,7 @@ class ProxyExtrasDBManager:
|
|||
check=True,
|
||||
timeout=60,
|
||||
stdout=f,
|
||||
env=_get_prisma_env()
|
||||
)
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.warning(f"Failed to generate migration diff: {e.stderr}")
|
||||
|
|
@ -245,7 +301,7 @@ class ProxyExtrasDBManager:
|
|||
logger.info("Running prisma db execute to apply the migration diff...")
|
||||
result = subprocess.run(
|
||||
[
|
||||
"prisma",
|
||||
_get_prisma_command(),
|
||||
"db",
|
||||
"execute",
|
||||
"--file",
|
||||
|
|
@ -257,6 +313,7 @@ class ProxyExtrasDBManager:
|
|||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=_get_prisma_env()
|
||||
)
|
||||
logger.info(f"prisma db execute stdout: {result.stdout}")
|
||||
logger.info("✅ Migration diff applied successfully")
|
||||
|
|
@ -274,11 +331,12 @@ class ProxyExtrasDBManager:
|
|||
try:
|
||||
logger.info(f"Resolving migration: {migration_name}")
|
||||
subprocess.run(
|
||||
["prisma", "migrate", "resolve", "--applied", migration_name],
|
||||
[_get_prisma_command(), "migrate", "resolve", "--applied", migration_name],
|
||||
timeout=60,
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=_get_prisma_env()
|
||||
)
|
||||
logger.debug(f"Resolved migration: {migration_name}")
|
||||
except subprocess.CalledProcessError as e:
|
||||
|
|
@ -312,11 +370,12 @@ class ProxyExtrasDBManager:
|
|||
try:
|
||||
# Set migrations directory for Prisma
|
||||
result = subprocess.run(
|
||||
["prisma", "migrate", "deploy"],
|
||||
[_get_prisma_command(), "migrate", "deploy"],
|
||||
timeout=60,
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=_get_prisma_env()
|
||||
)
|
||||
logger.info(f"prisma migrate deploy stdout: {result.stdout}")
|
||||
|
||||
|
|
@ -344,7 +403,7 @@ class ProxyExtrasDBManager:
|
|||
# Mark the failed migration as rolled back
|
||||
subprocess.run(
|
||||
[
|
||||
"prisma",
|
||||
_get_prisma_command(),
|
||||
"migrate",
|
||||
"resolve",
|
||||
"--rolled-back",
|
||||
|
|
@ -354,6 +413,7 @@ class ProxyExtrasDBManager:
|
|||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env=_get_prisma_env()
|
||||
)
|
||||
logger.info(
|
||||
f"✅ Migration {failed_migration} marked as rolled back... retrying"
|
||||
|
|
@ -450,7 +510,7 @@ class ProxyExtrasDBManager:
|
|||
else:
|
||||
# Use prisma db push with increased timeout
|
||||
subprocess.run(
|
||||
["prisma", "db", "push", "--accept-data-loss"],
|
||||
[_get_prisma_command(), "db", "push", "--accept-data-loss"],
|
||||
timeout=60,
|
||||
check=True,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -26,16 +26,6 @@ from typing import (
|
|||
)
|
||||
from litellm.types.integrations.datadog_llm_obs import DatadogLLMObsInitParams
|
||||
from litellm.types.integrations.datadog import DatadogInitParams
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache
|
||||
from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES
|
||||
from litellm.types.utils import (
|
||||
ImageObject,
|
||||
BudgetConfig,
|
||||
all_litellm_params,
|
||||
all_litellm_params as _litellm_completion_params,
|
||||
CredentialItem,
|
||||
PriorityReservationDict,
|
||||
) # maintain backwards compatibility for root param.
|
||||
from litellm._logging import (
|
||||
set_verbose,
|
||||
_turn_on_debug,
|
||||
|
|
@ -82,11 +72,6 @@ from litellm.constants import (
|
|||
DEFAULT_SOFT_BUDGET,
|
||||
DEFAULT_ALLOWED_FAILS,
|
||||
)
|
||||
from litellm.integrations.dotprompt import (
|
||||
global_prompt_manager,
|
||||
global_prompt_directory,
|
||||
set_global_prompt_directory,
|
||||
)
|
||||
from litellm.types.guardrails import GuardrailItem
|
||||
from litellm.types.secret_managers.main import (
|
||||
KeyManagementSystem,
|
||||
|
|
@ -96,11 +81,7 @@ from litellm.types.proxy.management_endpoints.ui_sso import (
|
|||
DefaultTeamSSOParams,
|
||||
LiteLLM_UpperboundKeyGenerateParams,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
StandardKeyGenerationConfig,
|
||||
LlmProviders,
|
||||
SearchProviders,
|
||||
)
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.types.utils import PriorityReservationSettings
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.logging_callback_manager import LoggingCallbackManager
|
||||
|
|
@ -285,7 +266,7 @@ disable_token_counter: bool = False
|
|||
disable_add_transform_inline_image_block: bool = False
|
||||
disable_add_user_agent_to_request_tags: bool = False
|
||||
extra_spend_tag_headers: Optional[List[str]] = None
|
||||
in_memory_llm_clients_cache: LLMClientCache = LLMClientCache()
|
||||
in_memory_llm_clients_cache: "LLMClientCache"
|
||||
safe_memory_mode: bool = False
|
||||
enable_azure_ad_token_refresh: Optional[bool] = False
|
||||
### DEFAULT AZURE API VERSION ###
|
||||
|
|
@ -293,9 +274,9 @@ AZURE_DEFAULT_API_VERSION = "2025-02-01-preview" # this is updated to the lates
|
|||
### DEFAULT WATSONX API VERSION ###
|
||||
WATSONX_DEFAULT_API_VERSION = "2024-03-13"
|
||||
### COHERE EMBEDDINGS DEFAULT TYPE ###
|
||||
COHERE_DEFAULT_EMBEDDING_INPUT_TYPE: COHERE_EMBEDDING_INPUT_TYPES = "search_document"
|
||||
COHERE_DEFAULT_EMBEDDING_INPUT_TYPE: "COHERE_EMBEDDING_INPUT_TYPES" = "search_document"
|
||||
### CREDENTIALS ###
|
||||
credential_list: List[CredentialItem] = []
|
||||
credential_list: List["CredentialItem"] = []
|
||||
### GUARDRAILS ###
|
||||
llamaguard_model_name: Optional[str] = None
|
||||
openai_moderations_model_name: Optional[str] = None
|
||||
|
|
@ -370,7 +351,7 @@ aws_sqs_callback_params: Optional[Dict] = None
|
|||
generic_logger_headers: Optional[Dict] = None
|
||||
default_key_generate_params: Optional[Dict] = None
|
||||
upperbound_key_generate_params: Optional[LiteLLM_UpperboundKeyGenerateParams] = None
|
||||
key_generation_settings: Optional[StandardKeyGenerationConfig] = None
|
||||
key_generation_settings: Optional["StandardKeyGenerationConfig"] = None
|
||||
default_internal_user_params: Optional[Dict] = None
|
||||
default_team_params: Optional[Union[DefaultTeamSSOParams, Dict]] = None
|
||||
default_team_settings: Optional[List] = None
|
||||
|
|
@ -379,7 +360,7 @@ default_max_internal_user_budget: Optional[float] = None
|
|||
max_internal_user_budget: Optional[float] = None
|
||||
max_ui_session_budget: Optional[float] = 10 # $10 USD budgets for UI Chat sessions
|
||||
internal_user_budget_duration: Optional[str] = None
|
||||
tag_budget_config: Optional[Dict[str, BudgetConfig]] = None
|
||||
tag_budget_config: Optional[Dict[str, "BudgetConfig"]] = None
|
||||
max_end_user_budget: Optional[float] = None
|
||||
max_end_user_budget_id: Optional[str] = None
|
||||
disable_end_user_cost_tracking: Optional[bool] = None
|
||||
|
|
@ -402,7 +383,9 @@ public_agent_groups: Optional[List[str]] = None
|
|||
# Old format: { "displayName": "url" } (for backward compatibility)
|
||||
public_model_groups_links: Dict[str, Union[str, Dict[str, Any]]] = {}
|
||||
#### REQUEST PRIORITIZATION #######
|
||||
priority_reservation: Optional[Dict[str, Union[float, PriorityReservationDict]]] = None
|
||||
priority_reservation: Optional[
|
||||
Dict[str, Union[float, "PriorityReservationDict"]]
|
||||
] = None
|
||||
priority_reservation_settings: "PriorityReservationSettings" = (
|
||||
PriorityReservationSettings()
|
||||
)
|
||||
|
|
@ -1471,7 +1454,6 @@ from . import rag
|
|||
|
||||
### CUSTOM LLMs ###
|
||||
from .types.llms.custom_llm import CustomLLMItem
|
||||
from .types.utils import GenericStreamingChunk
|
||||
|
||||
custom_provider_map: List[CustomLLMItem] = []
|
||||
_custom_providers: List[str] = (
|
||||
|
|
@ -1515,6 +1497,14 @@ if TYPE_CHECKING:
|
|||
from litellm.types.utils import ModelInfo as _ModelInfoType
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.caching.caching import Cache
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache
|
||||
from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES
|
||||
from litellm.types.utils import (
|
||||
BudgetConfig,
|
||||
CredentialItem,
|
||||
PriorityReservationDict,
|
||||
StandardKeyGenerationConfig,
|
||||
)
|
||||
|
||||
# Cost calculator functions
|
||||
cost_per_token: Callable[..., Tuple[float, float]]
|
||||
|
|
@ -1567,8 +1557,12 @@ def __getattr__(name: str) -> Any:
|
|||
LITELLM_LOGGING_NAMES,
|
||||
UTILS_NAMES,
|
||||
TOKEN_COUNTER_NAMES,
|
||||
LLM_CLIENT_CACHE_NAMES,
|
||||
BEDROCK_TYPES_NAMES,
|
||||
TYPES_UTILS_NAMES,
|
||||
CACHING_NAMES,
|
||||
HTTP_HANDLER_NAMES,
|
||||
DOTPROMPT_NAMES,
|
||||
)
|
||||
|
||||
# Lazy load cost_calculator functions
|
||||
|
|
@ -1591,6 +1585,21 @@ def __getattr__(name: str) -> Any:
|
|||
from ._lazy_imports import _lazy_import_token_counter
|
||||
return _lazy_import_token_counter(name)
|
||||
|
||||
# Lazy load Bedrock type aliases
|
||||
if name in BEDROCK_TYPES_NAMES:
|
||||
from ._lazy_imports import _lazy_import_bedrock_types
|
||||
return _lazy_import_bedrock_types(name)
|
||||
|
||||
# Lazy load common types.utils symbols
|
||||
if name in TYPES_UTILS_NAMES:
|
||||
from ._lazy_imports import _lazy_import_types_utils
|
||||
return _lazy_import_types_utils(name)
|
||||
|
||||
# Lazy load LLM client cache and its singleton
|
||||
if name in LLM_CLIENT_CACHE_NAMES:
|
||||
from ._lazy_imports import _lazy_import_llm_client_cache
|
||||
return _lazy_import_llm_client_cache(name)
|
||||
|
||||
# Lazy load caching classes
|
||||
if name in CACHING_NAMES:
|
||||
from ._lazy_imports import _lazy_import_caching
|
||||
|
|
@ -1602,6 +1611,12 @@ def __getattr__(name: str) -> Any:
|
|||
|
||||
return _lazy_import_http_handlers(name)
|
||||
|
||||
# Lazy load dotprompt integration globals
|
||||
if name in DOTPROMPT_NAMES:
|
||||
from ._lazy_imports import _lazy_import_dotprompt
|
||||
|
||||
return _lazy_import_dotprompt(name)
|
||||
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,10 +1,31 @@
|
|||
from typing import Any, cast
|
||||
from typing import Any, Optional, cast
|
||||
import sys
|
||||
|
||||
def _get_litellm_globals() -> dict:
|
||||
"""Helper to get the globals dictionary of the litellm module."""
|
||||
return sys.modules["litellm"].__dict__
|
||||
|
||||
# Lazy loader for default encoding to avoid importing tiktoken at module import time
|
||||
_default_encoding: Optional[Any] = None
|
||||
|
||||
|
||||
def _get_default_encoding() -> Any:
|
||||
"""
|
||||
Lazily load and cache the default OpenAI encoding.
|
||||
|
||||
This avoids importing `litellm.litellm_core_utils.default_encoding` (and thus tiktoken)
|
||||
at `litellm` import time. The encoding is cached after the first import.
|
||||
|
||||
This is used internally by utils.py functions that need the encoding but shouldn't
|
||||
trigger its import during module load.
|
||||
"""
|
||||
global _default_encoding
|
||||
if _default_encoding is None:
|
||||
from litellm.litellm_core_utils.default_encoding import encoding
|
||||
|
||||
_default_encoding = encoding
|
||||
return _default_encoding
|
||||
|
||||
# Cost calculator names that support lazy loading via _lazy_import_cost_calculator
|
||||
COST_CALCULATOR_NAMES = (
|
||||
"completion_cost",
|
||||
|
|
@ -39,6 +60,31 @@ TOKEN_COUNTER_NAMES = (
|
|||
"get_modified_max_tokens",
|
||||
)
|
||||
|
||||
# LLM client cache names that support lazy loading via _lazy_import_llm_client_cache
|
||||
LLM_CLIENT_CACHE_NAMES = (
|
||||
"LLMClientCache",
|
||||
"in_memory_llm_clients_cache",
|
||||
)
|
||||
|
||||
# Bedrock type names that support lazy loading via _lazy_import_bedrock_types
|
||||
BEDROCK_TYPES_NAMES = (
|
||||
"COHERE_EMBEDDING_INPUT_TYPES",
|
||||
)
|
||||
|
||||
# Common types from litellm.types.utils that support lazy loading via
|
||||
# _lazy_import_types_utils
|
||||
TYPES_UTILS_NAMES = (
|
||||
"ImageObject",
|
||||
"BudgetConfig",
|
||||
"all_litellm_params",
|
||||
"_litellm_completion_params",
|
||||
"CredentialItem",
|
||||
"PriorityReservationDict",
|
||||
"StandardKeyGenerationConfig",
|
||||
"SearchProviders",
|
||||
"GenericStreamingChunk",
|
||||
)
|
||||
|
||||
# Caching / cache classes that support lazy loading via _lazy_import_caching
|
||||
CACHING_NAMES = (
|
||||
"Cache",
|
||||
|
|
@ -53,6 +99,13 @@ HTTP_HANDLER_NAMES = (
|
|||
"module_level_client",
|
||||
)
|
||||
|
||||
# Dotprompt integration names that support lazy loading via _lazy_import_dotprompt
|
||||
DOTPROMPT_NAMES = (
|
||||
"global_prompt_manager",
|
||||
"global_prompt_directory",
|
||||
"set_global_prompt_directory",
|
||||
)
|
||||
|
||||
# Lazy import for utils module - imports only the requested item by name.
|
||||
# Note: PLR0915 (too many statements) is suppressed because the many if statements
|
||||
# are intentional - each attribute is imported individually only when requested,
|
||||
|
|
@ -299,6 +352,88 @@ def _lazy_import_token_counter(name: str) -> Any:
|
|||
raise AttributeError(f"Token counter lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_bedrock_types(name: str) -> Any:
|
||||
"""Lazy import for Bedrock type aliases."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
||||
if name == "COHERE_EMBEDDING_INPUT_TYPES":
|
||||
from litellm.types.llms.bedrock import (
|
||||
COHERE_EMBEDDING_INPUT_TYPES as _COHERE_EMBEDDING_INPUT_TYPES,
|
||||
)
|
||||
|
||||
_globals["COHERE_EMBEDDING_INPUT_TYPES"] = _COHERE_EMBEDDING_INPUT_TYPES
|
||||
return _COHERE_EMBEDDING_INPUT_TYPES
|
||||
|
||||
raise AttributeError(f"Bedrock types lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_types_utils(name: str) -> Any:
|
||||
"""Lazy import for common types and constants from litellm.types.utils."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
||||
if name == "ImageObject":
|
||||
from .types.utils import ImageObject as _ImageObject
|
||||
|
||||
_globals["ImageObject"] = _ImageObject
|
||||
return _ImageObject
|
||||
|
||||
if name == "BudgetConfig":
|
||||
from .types.utils import BudgetConfig as _BudgetConfig
|
||||
|
||||
_globals["BudgetConfig"] = _BudgetConfig
|
||||
return _BudgetConfig
|
||||
|
||||
if name == "all_litellm_params":
|
||||
from .types.utils import all_litellm_params as _all_litellm_params
|
||||
|
||||
_globals["all_litellm_params"] = _all_litellm_params
|
||||
return _all_litellm_params
|
||||
|
||||
if name == "_litellm_completion_params":
|
||||
from .types.utils import all_litellm_params as _all_litellm_params
|
||||
|
||||
_globals["_litellm_completion_params"] = _all_litellm_params
|
||||
return _all_litellm_params
|
||||
|
||||
if name == "CredentialItem":
|
||||
from .types.utils import CredentialItem as _CredentialItem
|
||||
|
||||
_globals["CredentialItem"] = _CredentialItem
|
||||
return _CredentialItem
|
||||
|
||||
if name == "PriorityReservationDict":
|
||||
from .types.utils import (
|
||||
PriorityReservationDict as _PriorityReservationDict,
|
||||
)
|
||||
|
||||
_globals["PriorityReservationDict"] = _PriorityReservationDict
|
||||
return _PriorityReservationDict
|
||||
|
||||
if name == "StandardKeyGenerationConfig":
|
||||
from .types.utils import (
|
||||
StandardKeyGenerationConfig as _StandardKeyGenerationConfig,
|
||||
)
|
||||
|
||||
_globals["StandardKeyGenerationConfig"] = _StandardKeyGenerationConfig
|
||||
return _StandardKeyGenerationConfig
|
||||
|
||||
if name == "SearchProviders":
|
||||
from .types.utils import SearchProviders as _SearchProviders
|
||||
|
||||
_globals["SearchProviders"] = _SearchProviders
|
||||
return _SearchProviders
|
||||
|
||||
if name == "GenericStreamingChunk":
|
||||
from .types.utils import (
|
||||
GenericStreamingChunk as _GenericStreamingChunk,
|
||||
)
|
||||
|
||||
_globals["GenericStreamingChunk"] = _GenericStreamingChunk
|
||||
return _GenericStreamingChunk
|
||||
|
||||
raise AttributeError(f"Types utils lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_caching(name: str) -> Any:
|
||||
"""Lazy import for caching module classes."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
|
@ -330,6 +465,28 @@ def _lazy_import_caching(name: str) -> Any:
|
|||
raise AttributeError(f"Caching lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_llm_client_cache(name: str) -> Any:
|
||||
"""Lazy import for LLM client cache class and singleton."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
||||
if name == "LLMClientCache":
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache as _LLMClientCache
|
||||
|
||||
_globals["LLMClientCache"] = _LLMClientCache
|
||||
return _LLMClientCache
|
||||
|
||||
if name == "in_memory_llm_clients_cache":
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache as _LLMClientCache
|
||||
|
||||
instance = _LLMClientCache()
|
||||
# Only populate the requested singleton name to keep lazy-import
|
||||
# semantics consistent with other helpers (no extra symbols).
|
||||
_globals["in_memory_llm_clients_cache"] = instance
|
||||
return instance
|
||||
|
||||
raise AttributeError(f"LLM client cache lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_litellm_logging(name: str) -> Any:
|
||||
"""Lazy import for litellm_logging module."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
|
@ -375,4 +532,35 @@ def _lazy_import_http_handlers(name: str) -> Any:
|
|||
_globals["module_level_client"] = sync_client
|
||||
return sync_client
|
||||
|
||||
raise AttributeError(f"HTTP handlers lazy import: unknown attribute {name!r}")
|
||||
raise AttributeError(f"HTTP handlers lazy import: unknown attribute {name!r}")
|
||||
|
||||
|
||||
def _lazy_import_dotprompt(name: str) -> Any:
|
||||
"""Lazy import for dotprompt integration globals."""
|
||||
_globals = _get_litellm_globals()
|
||||
|
||||
if name == "global_prompt_manager":
|
||||
from litellm.integrations.dotprompt import (
|
||||
global_prompt_manager as _global_prompt_manager,
|
||||
)
|
||||
|
||||
_globals["global_prompt_manager"] = _global_prompt_manager
|
||||
return _global_prompt_manager
|
||||
|
||||
if name == "global_prompt_directory":
|
||||
from litellm.integrations.dotprompt import (
|
||||
global_prompt_directory as _global_prompt_directory,
|
||||
)
|
||||
|
||||
_globals["global_prompt_directory"] = _global_prompt_directory
|
||||
return _global_prompt_directory
|
||||
|
||||
if name == "set_global_prompt_directory":
|
||||
from litellm.integrations.dotprompt import (
|
||||
set_global_prompt_directory as _set_global_prompt_directory,
|
||||
)
|
||||
|
||||
_globals["set_global_prompt_directory"] = _set_global_prompt_directory
|
||||
return _set_global_prompt_directory
|
||||
|
||||
raise AttributeError(f"Dotprompt lazy import: unknown attribute {name!r}")
|
||||
|
|
@ -6,12 +6,11 @@ This module provides fake streaming by converting non-streaming responses into s
|
|||
"""
|
||||
|
||||
import asyncio
|
||||
from typing import Any, AsyncIterator, Dict
|
||||
from typing import Any, AsyncIterator, Dict, cast
|
||||
from uuid import uuid4
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, get_async_httpx_client
|
||||
|
||||
|
||||
class PydanticAITransformation:
|
||||
|
|
@ -78,7 +77,7 @@ class PydanticAITransformation:
|
|||
|
||||
@staticmethod
|
||||
async def _poll_for_completion(
|
||||
client: httpx.AsyncClient,
|
||||
client: AsyncHTTPHandler,
|
||||
endpoint: str,
|
||||
task_id: str,
|
||||
request_id: str,
|
||||
|
|
@ -179,34 +178,37 @@ class PydanticAITransformation:
|
|||
f"Pydantic AI: Sending non-streaming request to {endpoint}"
|
||||
)
|
||||
|
||||
# Send request to Pydantic AI agent
|
||||
async with httpx.AsyncClient(timeout=timeout) as client:
|
||||
response = await client.post(
|
||||
endpoint,
|
||||
json=a2a_request,
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
response.raise_for_status()
|
||||
response_data = response.json()
|
||||
|
||||
# Check if task is already completed
|
||||
result = response_data.get("result", {})
|
||||
status = result.get("status", {})
|
||||
state = status.get("state", "")
|
||||
|
||||
if state != "completed":
|
||||
# Need to poll for completion
|
||||
task_id = result.get("id")
|
||||
if task_id:
|
||||
verbose_logger.info(
|
||||
f"Pydantic AI: Task {task_id} submitted, polling for completion..."
|
||||
)
|
||||
response_data = await PydanticAITransformation._poll_for_completion(
|
||||
client=client,
|
||||
endpoint=endpoint,
|
||||
task_id=task_id,
|
||||
request_id=request_id,
|
||||
)
|
||||
# Send request to Pydantic AI agent using shared async HTTP client
|
||||
client = get_async_httpx_client(
|
||||
llm_provider=cast(Any, "pydantic_ai_agent"),
|
||||
params={"timeout": timeout},
|
||||
)
|
||||
response = await client.post(
|
||||
endpoint,
|
||||
json=a2a_request,
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
response.raise_for_status()
|
||||
response_data = response.json()
|
||||
|
||||
# Check if task is already completed
|
||||
result = response_data.get("result", {})
|
||||
status = result.get("status", {})
|
||||
state = status.get("state", "")
|
||||
|
||||
if state != "completed":
|
||||
# Need to poll for completion
|
||||
task_id = result.get("id")
|
||||
if task_id:
|
||||
verbose_logger.info(
|
||||
f"Pydantic AI: Task {task_id} submitted, polling for completion..."
|
||||
)
|
||||
response_data = await PydanticAITransformation._poll_for_completion(
|
||||
client=client,
|
||||
endpoint=endpoint,
|
||||
task_id=task_id,
|
||||
request_id=request_id,
|
||||
)
|
||||
|
||||
verbose_logger.info(f"Pydantic AI: Received completed response for request_id={request_id}")
|
||||
|
||||
|
|
|
|||
|
|
@ -457,6 +457,24 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
raw_response.usage
|
||||
),
|
||||
)
|
||||
|
||||
# Preserve hidden params from the ResponsesAPIResponse, especially the headers
|
||||
# which contain important provider information like x-request-id
|
||||
raw_response_hidden_params = getattr(raw_response, "_hidden_params", {})
|
||||
if raw_response_hidden_params:
|
||||
if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None:
|
||||
model_response._hidden_params = {}
|
||||
# Merge the raw_response hidden params with model_response hidden params
|
||||
# Preserve existing keys in model_response but add/override with raw_response params
|
||||
for key, value in raw_response_hidden_params.items():
|
||||
if key == "additional_headers" and key in model_response._hidden_params:
|
||||
# Merge additional_headers to preserve both sets
|
||||
existing_additional_headers = model_response._hidden_params.get("additional_headers", {})
|
||||
merged_headers = {**value, **existing_additional_headers}
|
||||
model_response._hidden_params[key] = merged_headers
|
||||
else:
|
||||
model_response._hidden_params[key] = value
|
||||
|
||||
return model_response
|
||||
|
||||
def get_model_response_iterator(
|
||||
|
|
|
|||
|
|
@ -16,7 +16,6 @@ from typing import (
|
|||
from pydantic import BaseModel
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER
|
||||
from litellm.types.integrations.argilla import ArgillaItem
|
||||
from litellm.types.llms.openai import AllMessageValues, ChatCompletionRequest
|
||||
|
|
@ -33,6 +32,7 @@ from litellm.types.utils import (
|
|||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.caching.caching import DualCache
|
||||
from opentelemetry.trace import Span as _Span
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
|
@ -334,7 +334,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac
|
|||
async def async_pre_call_hook(
|
||||
self,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
cache: DualCache,
|
||||
cache: "DualCache",
|
||||
data: dict,
|
||||
call_type: CallTypesLiteral,
|
||||
) -> Optional[
|
||||
|
|
|
|||
|
|
@ -430,6 +430,18 @@ def convert_to_model_response_object( # noqa: PLR0915
|
|||
|
||||
if hidden_params is None:
|
||||
hidden_params = {}
|
||||
|
||||
# Preserve existing additional_headers if they contain important provider headers
|
||||
# For responses API, additional_headers may already be set with LLM provider headers
|
||||
existing_additional_headers = hidden_params.get("additional_headers", {})
|
||||
if existing_additional_headers and _response_headers is None:
|
||||
# Keep existing headers when _response_headers is None (responses API case)
|
||||
additional_headers = existing_additional_headers
|
||||
else:
|
||||
# Merge new headers with existing ones
|
||||
if existing_additional_headers:
|
||||
additional_headers.update(existing_additional_headers)
|
||||
|
||||
hidden_params["additional_headers"] = additional_headers
|
||||
|
||||
### CHECK IF ERROR IN RESPONSE ### - openrouter returns these in the dictionary
|
||||
|
|
|
|||
|
|
@ -340,7 +340,7 @@ class AnthropicChatCompletion(BaseLLM):
|
|||
data = config.transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
optional_params={**optional_params, "is_vertex_request": is_vertex_request},
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -942,6 +942,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
self, headers: dict, optional_params: dict
|
||||
) -> dict:
|
||||
"""Update headers with optional anthropic beta."""
|
||||
|
||||
# Skip adding beta headers for Vertex requests
|
||||
# Vertex AI handles these headers differently
|
||||
is_vertex_request = optional_params.get("is_vertex_request", False)
|
||||
if is_vertex_request:
|
||||
return headers
|
||||
|
||||
_tools = optional_params.get("tools", [])
|
||||
for tool in _tools:
|
||||
|
|
@ -1067,6 +1073,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
):
|
||||
optional_params["metadata"] = {"user_id": _litellm_metadata["user_id"]}
|
||||
|
||||
# Remove internal LiteLLM parameters that should not be sent to Anthropic API
|
||||
optional_params.pop("is_vertex_request", None)
|
||||
|
||||
data = {
|
||||
"model": model,
|
||||
"messages": anthropic_messages,
|
||||
|
|
|
|||
|
|
@ -108,6 +108,27 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
stream=stream,
|
||||
)
|
||||
|
||||
def _remove_ttl_from_cache_control(
|
||||
self, anthropic_messages_request: Dict
|
||||
) -> None:
|
||||
"""
|
||||
Remove `ttl` field from cache_control in messages.
|
||||
Bedrock doesn't support the ttl field in cache_control.
|
||||
|
||||
Args:
|
||||
anthropic_messages_request: The request dictionary to modify in-place
|
||||
"""
|
||||
if "messages" in anthropic_messages_request:
|
||||
for message in anthropic_messages_request["messages"]:
|
||||
if isinstance(message, dict) and "content" in message:
|
||||
content = message["content"]
|
||||
if isinstance(content, list):
|
||||
for item in content:
|
||||
if isinstance(item, dict) and "cache_control" in item:
|
||||
cache_control = item["cache_control"]
|
||||
if isinstance(cache_control, dict) and "ttl" in cache_control:
|
||||
cache_control.pop("ttl", None)
|
||||
|
||||
def transform_anthropic_messages_request(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -141,8 +162,11 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
# 3. `model` is not allowed in request body for bedrock invoke
|
||||
if "model" in anthropic_messages_request:
|
||||
anthropic_messages_request.pop("model", None)
|
||||
|
||||
# 4. Remove `ttl` field from cache_control in messages (Bedrock doesn't support it)
|
||||
self._remove_ttl_from_cache_control(anthropic_messages_request)
|
||||
|
||||
# 4. AUTO-INJECT beta headers based on features used
|
||||
# 5. AUTO-INJECT beta headers based on features used
|
||||
anthropic_model_info = AnthropicModelInfo()
|
||||
tools = anthropic_messages_optional_request_params.get("tools")
|
||||
messages_typed = cast(List[AllMessageValues], messages)
|
||||
|
|
|
|||
|
|
@ -1153,7 +1153,17 @@ def get_async_httpx_client(
|
|||
pass
|
||||
|
||||
_cache_key_name = "async_httpx_client" + _params_key_name + llm_provider
|
||||
_cached_client = litellm.in_memory_llm_clients_cache.get_cache(_cache_key_name)
|
||||
|
||||
# Lazily initialize the global in-memory client cache to avoid relying on
|
||||
# litellm globals being fully populated during import time.
|
||||
cache = getattr(litellm, "in_memory_llm_clients_cache", None)
|
||||
if cache is None:
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache
|
||||
|
||||
cache = LLMClientCache()
|
||||
setattr(litellm, "in_memory_llm_clients_cache", cache)
|
||||
|
||||
_cached_client = cache.get_cache(_cache_key_name)
|
||||
if _cached_client:
|
||||
return _cached_client
|
||||
|
||||
|
|
@ -1166,7 +1176,7 @@ def get_async_httpx_client(
|
|||
shared_session=shared_session,
|
||||
)
|
||||
|
||||
litellm.in_memory_llm_clients_cache.set_cache(
|
||||
cache.set_cache(
|
||||
key=_cache_key_name,
|
||||
value=_new_client,
|
||||
ttl=_DEFAULT_TTL_FOR_HTTPX_CLIENTS,
|
||||
|
|
@ -1191,7 +1201,16 @@ def _get_httpx_client(params: Optional[dict] = None) -> HTTPHandler:
|
|||
|
||||
_cache_key_name = "httpx_client" + _params_key_name
|
||||
|
||||
_cached_client = litellm.in_memory_llm_clients_cache.get_cache(_cache_key_name)
|
||||
# Lazily initialize the global in-memory client cache to avoid relying on
|
||||
# litellm globals being fully populated during import time.
|
||||
cache = getattr(litellm, "in_memory_llm_clients_cache", None)
|
||||
if cache is None:
|
||||
from litellm.caching.llm_caching_handler import LLMClientCache
|
||||
|
||||
cache = LLMClientCache()
|
||||
setattr(litellm, "in_memory_llm_clients_cache", cache)
|
||||
|
||||
_cached_client = cache.get_cache(_cache_key_name)
|
||||
if _cached_client:
|
||||
return _cached_client
|
||||
|
||||
|
|
@ -1200,7 +1219,7 @@ def _get_httpx_client(params: Optional[dict] = None) -> HTTPHandler:
|
|||
else:
|
||||
_new_client = HTTPHandler(timeout=httpx.Timeout(timeout=600.0, connect=5.0))
|
||||
|
||||
litellm.in_memory_llm_clients_cache.set_cache(
|
||||
cache.set_cache(
|
||||
key=_cache_key_name,
|
||||
value=_new_client,
|
||||
ttl=_DEFAULT_TTL_FOR_HTTPX_CLIENTS,
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ from pydantic import BaseModel
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.core_helpers import process_response_headers
|
||||
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
|
||||
_safe_convert_created_field,
|
||||
)
|
||||
|
|
@ -15,7 +16,7 @@ from litellm.types.llms.openai import *
|
|||
from litellm.types.responses.main import *
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.litellm_core_utils.core_helpers import process_response_headers
|
||||
|
||||
from ..common_utils import OpenAIError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -181,6 +182,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
)
|
||||
response = ResponsesAPIResponse.model_construct(**raw_response_json)
|
||||
|
||||
# Store processed headers in additional_headers so they get returned to the client
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -14,5 +14,9 @@
|
|||
"helicone": {
|
||||
"base_url": "https://ai-gateway.helicone.ai/",
|
||||
"api_key_env": "HELICONE_API_KEY"
|
||||
},
|
||||
"veniceai": {
|
||||
"base_url": "https://api.venice.ai/api/v1",
|
||||
"api_key_env": "VENICE_AI_API_KEY"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
13
litellm/llms/vertex_ai/agent_engine/__init__.py
Normal file
13
litellm/llms/vertex_ai/agent_engine/__init__.py
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
"""
|
||||
Vertex AI Agent Engine (Reasoning Engines) Provider
|
||||
|
||||
Supports Vertex AI Reasoning Engines via the :query and :streamQuery endpoints.
|
||||
"""
|
||||
|
||||
from litellm.llms.vertex_ai.agent_engine.transformation import (
|
||||
VertexAgentEngineConfig,
|
||||
VertexAgentEngineError,
|
||||
)
|
||||
|
||||
__all__ = ["VertexAgentEngineConfig", "VertexAgentEngineError"]
|
||||
|
||||
90
litellm/llms/vertex_ai/agent_engine/sse_iterator.py
Normal file
90
litellm/llms/vertex_ai/agent_engine/sse_iterator.py
Normal file
|
|
@ -0,0 +1,90 @@
|
|||
"""
|
||||
SSE Stream Iterator for Vertex AI Agent Engine.
|
||||
|
||||
Handles Server-Sent Events (SSE) streaming responses from Vertex AI Reasoning Engines.
|
||||
"""
|
||||
|
||||
from typing import Any, Union
|
||||
|
||||
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
|
||||
from litellm.types.llms.openai import ChatCompletionUsageBlock
|
||||
from litellm.types.utils import (
|
||||
Delta,
|
||||
GenericStreamingChunk,
|
||||
ModelResponseStream,
|
||||
StreamingChoices,
|
||||
)
|
||||
|
||||
|
||||
class VertexAgentEngineResponseIterator(BaseModelResponseIterator):
|
||||
"""
|
||||
Iterator for Vertex Agent Engine SSE streaming responses.
|
||||
|
||||
Uses BaseModelResponseIterator which handles sync/async iteration.
|
||||
We just need to implement chunk_parser to parse Vertex Agent Engine response format.
|
||||
"""
|
||||
|
||||
def __init__(self, streaming_response: Any, sync_stream: bool) -> None:
|
||||
super().__init__(streaming_response=streaming_response, sync_stream=sync_stream)
|
||||
|
||||
def chunk_parser(
|
||||
self, chunk: dict
|
||||
) -> Union[GenericStreamingChunk, ModelResponseStream]:
|
||||
"""
|
||||
Parse a Vertex Agent Engine response chunk into ModelResponseStream.
|
||||
|
||||
Vertex Agent Engine response format:
|
||||
{
|
||||
"content": {
|
||||
"parts": [{"text": "..."}],
|
||||
"role": "model"
|
||||
},
|
||||
"finish_reason": "STOP",
|
||||
"usage_metadata": {
|
||||
"prompt_token_count": 100,
|
||||
"candidates_token_count": 50,
|
||||
"total_token_count": 150
|
||||
}
|
||||
}
|
||||
"""
|
||||
# Extract text from content.parts
|
||||
text = None
|
||||
content = chunk.get("content", {})
|
||||
parts = content.get("parts", [])
|
||||
for part in parts:
|
||||
if isinstance(part, dict) and "text" in part:
|
||||
text = part["text"]
|
||||
break
|
||||
|
||||
# Extract finish_reason
|
||||
finish_reason = None
|
||||
raw_finish_reason = chunk.get("finish_reason")
|
||||
if raw_finish_reason == "STOP":
|
||||
finish_reason = "stop"
|
||||
elif raw_finish_reason:
|
||||
finish_reason = raw_finish_reason.lower()
|
||||
|
||||
# Extract usage from usage_metadata
|
||||
usage = None
|
||||
usage_metadata = chunk.get("usage_metadata", {})
|
||||
if usage_metadata:
|
||||
usage = ChatCompletionUsageBlock(
|
||||
prompt_tokens=usage_metadata.get("prompt_token_count", 0),
|
||||
completion_tokens=usage_metadata.get("candidates_token_count", 0),
|
||||
total_tokens=usage_metadata.get("total_token_count", 0),
|
||||
)
|
||||
|
||||
# Return ModelResponseStream (OpenAI-compatible chunk)
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=finish_reason,
|
||||
index=0,
|
||||
delta=Delta(
|
||||
content=text,
|
||||
role="assistant" if text else None,
|
||||
),
|
||||
)
|
||||
],
|
||||
usage=usage,
|
||||
)
|
||||
508
litellm/llms/vertex_ai/agent_engine/transformation.py
Normal file
508
litellm/llms/vertex_ai/agent_engine/transformation.py
Normal file
|
|
@ -0,0 +1,508 @@
|
|||
"""
|
||||
Transformation for Vertex AI Agent Engine (Reasoning Engines)
|
||||
|
||||
Handles the transformation between LiteLLM's OpenAI-compatible format and
|
||||
Vertex AI Reasoning Engine's API format.
|
||||
|
||||
API Reference:
|
||||
- :query endpoint - for session management (create, get, list, delete)
|
||||
- :streamQuery endpoint - for actual queries (stream_query method)
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm._uuid import uuid
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
convert_content_list_to_str,
|
||||
)
|
||||
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
|
||||
from litellm.llms.vertex_ai.agent_engine.sse_iterator import (
|
||||
VertexAgentEngineResponseIterator,
|
||||
)
|
||||
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
LiteLLMLoggingObj = _LiteLLMLoggingObj
|
||||
else:
|
||||
LiteLLMLoggingObj = Any
|
||||
HTTPHandler = Any
|
||||
AsyncHTTPHandler = Any
|
||||
CustomStreamWrapper = Any
|
||||
|
||||
|
||||
class VertexAgentEngineError(BaseLLMException):
|
||||
"""Exception for Vertex Agent Engine errors."""
|
||||
|
||||
def __init__(self, status_code: int, message: str):
|
||||
self.status_code = status_code
|
||||
self.message = message
|
||||
super().__init__(message=message, status_code=status_code)
|
||||
|
||||
|
||||
class VertexAgentEngineConfig(BaseConfig, VertexBase):
|
||||
"""
|
||||
Configuration for Vertex AI Agent Engine (Reasoning Engines).
|
||||
|
||||
Model format: vertex_ai/agent_engine/<resource_id>
|
||||
Where resource_id is the numeric ID of the reasoning engine.
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
BaseConfig.__init__(self, **kwargs)
|
||||
VertexBase.__init__(self)
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> List[str]:
|
||||
"""Vertex Agent Engine has limited OpenAI compatible params."""
|
||||
return ["user"]
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
"""Map OpenAI params to Agent Engine params."""
|
||||
# Map 'user' to 'user_id' for session management
|
||||
if "user" in non_default_params:
|
||||
optional_params["user_id"] = non_default_params["user"]
|
||||
return optional_params
|
||||
|
||||
def _parse_model_string(self, model: str) -> Tuple[str, str]:
|
||||
"""
|
||||
Parse model string to extract resource ID.
|
||||
|
||||
Model format: agent_engine/<project_number>/<location>/<engine_id>
|
||||
Or: agent_engine/<engine_id> (uses default project/location)
|
||||
|
||||
Returns: (resource_path, engine_id)
|
||||
"""
|
||||
# Remove 'agent_engine/' prefix if present
|
||||
if model.startswith("agent_engine/"):
|
||||
model = model[len("agent_engine/") :]
|
||||
|
||||
# Check if it's a full resource path
|
||||
if model.startswith("projects/"):
|
||||
# Full path: projects/123/locations/us-central1/reasoningEngines/456
|
||||
return model, model.split("/")[-1]
|
||||
|
||||
# Just the engine ID
|
||||
return model, model
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: Optional[str],
|
||||
api_key: Optional[str],
|
||||
model: str,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
stream: Optional[bool] = None,
|
||||
) -> str:
|
||||
"""
|
||||
Get the complete URL for the request.
|
||||
|
||||
For Vertex Agent Engine:
|
||||
- Non-streaming: :query endpoint (for session management)
|
||||
- Streaming: :streamQuery endpoint (for actual queries)
|
||||
"""
|
||||
resource_path, engine_id = self._parse_model_string(model)
|
||||
|
||||
# Get project and location from litellm_params or environment
|
||||
vertex_project = self.safe_get_vertex_ai_project(litellm_params)
|
||||
vertex_location = self.safe_get_vertex_ai_location(litellm_params) or "us-central1"
|
||||
|
||||
# Build the full resource path if only engine_id was provided
|
||||
if not resource_path.startswith("projects/"):
|
||||
if not vertex_project:
|
||||
raise ValueError(
|
||||
"vertex_project is required for Vertex Agent Engine. "
|
||||
"Set via litellm_params['vertex_project'] or VERTEXAI_PROJECT env var."
|
||||
)
|
||||
resource_path = f"projects/{vertex_project}/locations/{vertex_location}/reasoningEngines/{engine_id}"
|
||||
|
||||
# Build the base URL
|
||||
base_url = f"https://{vertex_location}-aiplatform.googleapis.com"
|
||||
|
||||
# Always use :streamQuery endpoint for actual queries
|
||||
# The :query endpoint only supports session management methods
|
||||
# (create_session, get_session, list_sessions, delete_session, etc.)
|
||||
endpoint = f"{base_url}/v1beta1/{resource_path}:streamQuery"
|
||||
|
||||
verbose_logger.debug(f"Vertex Agent Engine URL: {endpoint}")
|
||||
return endpoint
|
||||
|
||||
def _get_auth_headers(
|
||||
self,
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
) -> Dict[str, str]:
|
||||
"""Get authentication headers using Google Cloud credentials."""
|
||||
vertex_credentials = self.safe_get_vertex_ai_credentials(litellm_params)
|
||||
vertex_project = self.safe_get_vertex_ai_project(litellm_params)
|
||||
|
||||
# Get access token using VertexBase
|
||||
access_token, project_id = self.get_access_token(
|
||||
credentials=vertex_credentials,
|
||||
project_id=vertex_project,
|
||||
)
|
||||
|
||||
verbose_logger.debug(f"Vertex Agent Engine: Authenticated for project {project_id}")
|
||||
|
||||
return {
|
||||
"Authorization": f"Bearer {access_token}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
|
||||
def _get_user_id(self, optional_params: dict) -> str:
|
||||
"""Get or generate user ID for session management."""
|
||||
user_id = optional_params.get("user_id") or optional_params.get("user")
|
||||
if user_id:
|
||||
return user_id
|
||||
# Generate a user ID
|
||||
return f"litellm-user-{str(uuid.uuid4())[:8]}"
|
||||
|
||||
def _get_session_id(self, optional_params: dict) -> Optional[str]:
|
||||
"""Get session ID if provided."""
|
||||
return optional_params.get("session_id")
|
||||
|
||||
def transform_request(
|
||||
self,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
headers: dict,
|
||||
) -> dict:
|
||||
"""
|
||||
Transform the request to Vertex Agent Engine format.
|
||||
|
||||
The API expects:
|
||||
{
|
||||
"class_method": "stream_query",
|
||||
"input": {
|
||||
"message": "...",
|
||||
"user_id": "...",
|
||||
"session_id": "..." (optional)
|
||||
}
|
||||
}
|
||||
"""
|
||||
# Use the last message content as the prompt
|
||||
prompt = convert_content_list_to_str(messages[-1])
|
||||
|
||||
# Get user_id and session_id
|
||||
user_id = self._get_user_id(optional_params)
|
||||
session_id = self._get_session_id(optional_params)
|
||||
|
||||
# Build the input
|
||||
input_data: Dict[str, Any] = {
|
||||
"message": prompt,
|
||||
"user_id": user_id,
|
||||
}
|
||||
|
||||
if session_id:
|
||||
input_data["session_id"] = session_id
|
||||
|
||||
# Build the request payload
|
||||
# Note: stream_query is used for both streaming and non-streaming
|
||||
# The difference is the endpoint (:streamQuery vs :query)
|
||||
payload = {
|
||||
"class_method": "stream_query",
|
||||
"input": input_data,
|
||||
}
|
||||
|
||||
verbose_logger.debug(f"Vertex Agent Engine payload: {payload}")
|
||||
return payload
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict,
|
||||
model: str,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
api_key: Optional[str] = None,
|
||||
api_base: Optional[str] = None,
|
||||
) -> dict:
|
||||
"""Validate environment and set up authentication headers."""
|
||||
auth_headers = self._get_auth_headers(optional_params, litellm_params)
|
||||
headers.update(auth_headers)
|
||||
return headers
|
||||
|
||||
def _extract_text_from_response(self, response_data: dict) -> str:
|
||||
"""Extract text content from the response."""
|
||||
# Try to get from content.parts
|
||||
content = response_data.get("content", {})
|
||||
parts = content.get("parts", [])
|
||||
for part in parts:
|
||||
if "text" in part:
|
||||
return part["text"]
|
||||
|
||||
# Try actions.state_delta
|
||||
actions = response_data.get("actions", {})
|
||||
state_delta = actions.get("state_delta", {})
|
||||
for key, value in state_delta.items():
|
||||
if isinstance(value, str) and value:
|
||||
return value
|
||||
|
||||
return ""
|
||||
|
||||
def _calculate_usage(
|
||||
self, model: str, messages: List[AllMessageValues], content: str
|
||||
) -> Optional[Usage]:
|
||||
"""Calculate token usage using LiteLLM's token counter."""
|
||||
try:
|
||||
from litellm.utils import token_counter
|
||||
|
||||
prompt_tokens = token_counter(model="gpt-3.5-turbo", messages=messages)
|
||||
completion_tokens = token_counter(
|
||||
model="gpt-3.5-turbo", text=content, count_response_tokens=True
|
||||
)
|
||||
total_tokens = prompt_tokens + completion_tokens
|
||||
|
||||
return Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=total_tokens,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Failed to calculate token usage: {str(e)}")
|
||||
return None
|
||||
|
||||
def transform_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ModelResponse,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
request_data: dict,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
encoding: Any,
|
||||
api_key: Optional[str] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
) -> ModelResponse:
|
||||
"""
|
||||
Transform Vertex Agent Engine response to LiteLLM ModelResponse format.
|
||||
|
||||
The response is a streaming SSE format even for non-streaming requests.
|
||||
We need to collect all the chunks and extract the final response.
|
||||
"""
|
||||
try:
|
||||
content_type = raw_response.headers.get("content-type", "").lower()
|
||||
verbose_logger.debug(f"Vertex Agent Engine response Content-Type: {content_type}")
|
||||
|
||||
# Parse the SSE response
|
||||
response_text = raw_response.text
|
||||
verbose_logger.debug(f"Response (first 500 chars): {response_text[:500]}")
|
||||
|
||||
# Extract content from SSE stream
|
||||
content = ""
|
||||
for line in response_text.strip().split("\n"):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
try:
|
||||
data = json.loads(line)
|
||||
if isinstance(data, dict):
|
||||
text = self._extract_text_from_response(data)
|
||||
if text:
|
||||
content = text # Use the last non-empty text
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
# Create the message
|
||||
message = Message(content=content, role="assistant")
|
||||
|
||||
# Create choices
|
||||
choice = Choices(finish_reason="stop", index=0, message=message)
|
||||
|
||||
# Update model response
|
||||
model_response.choices = [choice]
|
||||
model_response.model = model
|
||||
|
||||
# Calculate usage
|
||||
calculated_usage = self._calculate_usage(model, messages, content)
|
||||
if calculated_usage:
|
||||
setattr(model_response, "usage", calculated_usage)
|
||||
|
||||
return model_response
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error processing Vertex Agent Engine response: {str(e)}")
|
||||
raise VertexAgentEngineError(
|
||||
message=f"Error processing response: {str(e)}",
|
||||
status_code=raw_response.status_code,
|
||||
)
|
||||
|
||||
def get_streaming_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
) -> VertexAgentEngineResponseIterator:
|
||||
"""Return a streaming iterator for SSE responses."""
|
||||
return VertexAgentEngineResponseIterator(
|
||||
streaming_response=raw_response.iter_lines(),
|
||||
sync_stream=True,
|
||||
)
|
||||
|
||||
def get_sync_custom_stream_wrapper(
|
||||
self,
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
api_base: str,
|
||||
headers: dict,
|
||||
data: dict,
|
||||
messages: list,
|
||||
client: Optional[Union[HTTPHandler, "AsyncHTTPHandler"]] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
signed_json_body: Optional[bytes] = None,
|
||||
) -> "CustomStreamWrapper":
|
||||
"""Get a CustomStreamWrapper for synchronous streaming."""
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
HTTPHandler,
|
||||
_get_httpx_client,
|
||||
)
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
if client is None or not isinstance(client, HTTPHandler):
|
||||
client = _get_httpx_client(params={})
|
||||
|
||||
# Avoid logging sensitive api_base directly
|
||||
verbose_logger.debug("Making sync streaming request to Vertex AI endpoint.")
|
||||
|
||||
# Make streaming request
|
||||
response = client.post(
|
||||
api_base,
|
||||
headers=headers,
|
||||
data=json.dumps(data),
|
||||
stream=True,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise VertexAgentEngineError(
|
||||
status_code=response.status_code, message=str(response.read())
|
||||
)
|
||||
|
||||
# Create iterator for SSE stream
|
||||
completion_stream = self.get_streaming_response(model=model, raw_response=response)
|
||||
|
||||
streaming_response = CustomStreamWrapper(
|
||||
completion_stream=completion_stream,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
# LOGGING
|
||||
logging_obj.post_call(
|
||||
input=messages,
|
||||
api_key="",
|
||||
original_response="first stream response received",
|
||||
additional_args={"complete_input_dict": data},
|
||||
)
|
||||
|
||||
return streaming_response
|
||||
|
||||
async def get_async_custom_stream_wrapper(
|
||||
self,
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
api_base: str,
|
||||
headers: dict,
|
||||
data: dict,
|
||||
messages: list,
|
||||
client: Optional["AsyncHTTPHandler"] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
signed_json_body: Optional[bytes] = None,
|
||||
) -> "CustomStreamWrapper":
|
||||
"""Get a CustomStreamWrapper for asynchronous streaming."""
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
AsyncHTTPHandler,
|
||||
get_async_httpx_client,
|
||||
)
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
if client is None or not isinstance(client, AsyncHTTPHandler):
|
||||
client = get_async_httpx_client(
|
||||
llm_provider=cast(Any, "vertex_ai"), params={}
|
||||
)
|
||||
|
||||
# Avoid logging sensitive api_base directly
|
||||
verbose_logger.debug("Making async streaming request to Vertex AI endpoint.")
|
||||
|
||||
# Make async streaming request
|
||||
response = await client.post(
|
||||
api_base,
|
||||
headers=headers,
|
||||
data=json.dumps(data),
|
||||
stream=True,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise VertexAgentEngineError(
|
||||
status_code=response.status_code, message=str(await response.aread())
|
||||
)
|
||||
|
||||
# Create iterator for SSE stream (async)
|
||||
completion_stream = VertexAgentEngineResponseIterator(
|
||||
streaming_response=response.aiter_lines(),
|
||||
sync_stream=False,
|
||||
)
|
||||
|
||||
streaming_response = CustomStreamWrapper(
|
||||
completion_stream=completion_stream,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
# LOGGING
|
||||
logging_obj.post_call(
|
||||
input=messages,
|
||||
api_key="",
|
||||
original_response="first stream response received",
|
||||
additional_args={"complete_input_dict": data},
|
||||
)
|
||||
|
||||
return streaming_response
|
||||
|
||||
@property
|
||||
def has_custom_stream_wrapper(self) -> bool:
|
||||
"""Indicates that this config has custom streaming support."""
|
||||
return True
|
||||
|
||||
@property
|
||||
def supports_stream_param_in_request_body(self) -> bool:
|
||||
"""Agent Engine does not allow passing `stream` in the request body."""
|
||||
return False
|
||||
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
|
||||
) -> BaseLLMException:
|
||||
return VertexAgentEngineError(status_code=status_code, message=error_message)
|
||||
|
||||
def should_fake_stream(
|
||||
self,
|
||||
model: Optional[str],
|
||||
stream: Optional[bool],
|
||||
custom_llm_provider: Optional[str] = None,
|
||||
) -> bool:
|
||||
"""Agent Engine always returns SSE streams, so we use real streaming."""
|
||||
return False
|
||||
|
||||
|
|
@ -5,7 +5,6 @@ from typing import Any, Dict, List, Literal, Optional, Set, Tuple, Union, get_ty
|
|||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm.utils import supports_response_schema, supports_system_messages
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import unpack_defs
|
||||
|
|
@ -14,6 +13,7 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
|||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.llms.vertex_ai import PartType, Schema
|
||||
from litellm.types.utils import TokenCountResponse
|
||||
from litellm.utils import supports_response_schema, supports_system_messages
|
||||
|
||||
|
||||
class VertexAIError(BaseLLMException):
|
||||
|
|
@ -36,6 +36,7 @@ class VertexAIModelRoute(str, Enum):
|
|||
MODEL_GARDEN = "model_garden"
|
||||
NON_GEMINI = "non_gemini"
|
||||
OPENAI_COMPATIBLE = "openai"
|
||||
AGENT_ENGINE = "agent_engine"
|
||||
|
||||
VERTEX_AI_MODEL_ROUTES = [f"{route.value}/" for route in VertexAIModelRoute]
|
||||
|
||||
|
|
@ -76,6 +77,10 @@ def get_vertex_ai_model_route(
|
|||
if litellm_params and litellm_params.get("base_model") is not None:
|
||||
if "gemini" in litellm_params["base_model"]:
|
||||
return VertexAIModelRoute.GEMINI
|
||||
|
||||
# Check for agent_engine models (Reasoning Engines)
|
||||
if "agent_engine/" in model:
|
||||
return VertexAIModelRoute.AGENT_ENGINE
|
||||
|
||||
# Check if numeric endpoint ID with custom api_base (PSC endpoint)
|
||||
# Route to GEMINI (HTTP path) to support PSC endpoints properly
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ from litellm.types.llms.openai import (
|
|||
AllMessageValues,
|
||||
OpenAIImageGenerationOptionalParams,
|
||||
)
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
from litellm.types.utils import ImageObject, ImageResponse, ImageUsage, ImageUsageInputTokensDetails
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj
|
||||
|
|
@ -234,6 +234,27 @@ class VertexAIGeminiImageGenerationConfig(BaseImageGenerationConfig, VertexLLM):
|
|||
|
||||
return request_body
|
||||
|
||||
def _transform_image_usage(self, usage: dict) -> ImageUsage:
|
||||
input_tokens_details = ImageUsageInputTokensDetails(
|
||||
image_tokens=0,
|
||||
text_tokens=0,
|
||||
)
|
||||
tokens_details = usage.get("promptTokensDetails", [])
|
||||
for details in tokens_details:
|
||||
if isinstance(details, dict) and (modality := details.get("modality")):
|
||||
token_count = details.get("tokenCount", 0)
|
||||
if modality == "TEXT":
|
||||
input_tokens_details.text_tokens += token_count
|
||||
elif modality == "IMAGE":
|
||||
input_tokens_details.image_tokens += token_count
|
||||
|
||||
return ImageUsage(
|
||||
input_tokens=usage.get("promptTokenCount", 0),
|
||||
input_tokens_details=input_tokens_details,
|
||||
output_tokens=usage.get("candidatesTokenCount", 0),
|
||||
total_tokens=usage.get("totalTokenCount", 0),
|
||||
)
|
||||
|
||||
def transform_image_generation_response(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -276,6 +297,9 @@ class VertexAIGeminiImageGenerationConfig(BaseImageGenerationConfig, VertexLLM):
|
|||
b64_json=inline_data["data"],
|
||||
url=None,
|
||||
))
|
||||
|
||||
if usage_metadata := response_data.get("usageMetadata", None):
|
||||
model_response.usage = self._transform_image_usage(usage_metadata)
|
||||
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -3242,6 +3242,37 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
timeout=timeout,
|
||||
client=client,
|
||||
)
|
||||
elif model_route == VertexAIModelRoute.AGENT_ENGINE:
|
||||
# Vertex AI Agent Engine (Reasoning Engines)
|
||||
from litellm.llms.vertex_ai.agent_engine.transformation import (
|
||||
VertexAgentEngineConfig,
|
||||
)
|
||||
|
||||
vertex_agent_engine_config = VertexAgentEngineConfig()
|
||||
|
||||
# Update litellm_params with vertex credentials
|
||||
litellm_params["vertex_project"] = vertex_ai_project
|
||||
litellm_params["vertex_location"] = vertex_ai_location
|
||||
litellm_params["vertex_credentials"] = vertex_credentials
|
||||
|
||||
model_response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
stream=stream,
|
||||
messages=messages,
|
||||
model_response=model_response,
|
||||
optional_params=new_params,
|
||||
litellm_params=litellm_params, # type: ignore
|
||||
encoding=encoding,
|
||||
api_key=None,
|
||||
api_base=api_base,
|
||||
logging_obj=logging,
|
||||
acompletion=acompletion,
|
||||
timeout=timeout,
|
||||
client=client,
|
||||
custom_llm_provider="vertex_ai",
|
||||
provider_config=vertex_agent_engine_config,
|
||||
headers=headers or {},
|
||||
)
|
||||
else: # VertexAIModelRoute.NON_GEMINI
|
||||
model_response = vertex_ai_non_gemini.completion(
|
||||
model=model,
|
||||
|
|
@ -4476,6 +4507,12 @@ def embedding( # noqa: PLR0915
|
|||
|
||||
if extra_headers is not None:
|
||||
optional_params["extra_headers"] = extra_headers
|
||||
|
||||
if encoding_format is not None:
|
||||
optional_params["encoding_format"] = encoding_format
|
||||
else:
|
||||
# Omiting causes openai sdk to add default value of "float"
|
||||
optional_params["encoding_format"] = None
|
||||
|
||||
api_version = None
|
||||
|
||||
|
|
|
|||
|
|
@ -30628,11 +30628,11 @@
|
|||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "embedding"
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b": {
|
||||
"fireworks_ai/accounts/fireworks/models/": {
|
||||
"max_tokens": 40960,
|
||||
"max_input_tokens": 40960,
|
||||
"max_output_tokens": 40960,
|
||||
"input_cost_per_token": 0.0,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 0.0,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "embedding"
|
||||
|
|
|
|||
|
|
@ -0,0 +1,5 @@
|
|||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 120 120">
|
||||
<path
|
||||
fill="#e92063"
|
||||
d="M 119.18,86.64 98.02,57.3 c 0,0 0,0 0,0 L 63.77,9.8 c -1.74,-2.4 -5.76,-2.4 -7.49,0 l -34.24,47.49 c 0,0 0,0 0,0 L 0.87,86.64 c -0.86,1.2 -1.1,2.73 -0.65,4.13 0.46,1.4 1.55,2.5 2.95,2.96 l 55.41,18.14 c 0,0 0,0 0.01,9e-4 0.46,0.15 0.94,0.23 1.43,0.23 0.49,0 0.97,-0.08 1.43,-0.23 0,0 0,0 0.01,0 L 116.87,93.73 c 1.4,-0.46 2.5,-1.55 2.95,-2.96 0.46,-1.4 0.22,-2.93 -0.65,-4.13 z m -59.15,-66.25 22.21,30.8 -20.77,-6.8 c -0.16,-0.05 -0.33,-0.04 -0.49,-0.08 -0.16,-0.04 -0.32,-0.06 -0.48,-0.08 -0.16,-0.02 -0.31,-0.08 -0.47,-0.08 -0.16,0 -0.31,0.06 -0.47,0.08 -0.17,0.02 -0.32,0.04 -0.48,0.08 -0.16,0.03 -0.33,0.03 -0.48,0.08 h 0 l -20.64,6.76 -0.13,0.04 22.21,-30.8 z m -31.38,43.52 24.18,-7.92 2.58,-0.84 V 101.12 L 12.06,86.92 Z m 36,37.2 V 55.15 l 26.76,8.76 16.59,23 z"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 880 B |
|
|
@ -605,13 +605,13 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
|
|||
"""
|
||||
Only raise exception for "BLOCKED" actions, not for "ANONYMIZED" actions.
|
||||
|
||||
If `self.mask_request_content` or `self.mask_response_content` is set to `True`, then use the output from the guardrail to mask the request or response content.
|
||||
If `self.mask_request_content` or `self.mask_response_content` is set to `True`,
|
||||
then use the output from the guardrail to mask the request or response content.
|
||||
|
||||
However, even with masking enabled, content with action="BLOCKED" should still
|
||||
raise an exception, only content with action="ANONYMIZED" should be masked.
|
||||
"""
|
||||
|
||||
# if user opted into masking, return False. since we'll use the masked output from the guardrail
|
||||
if self.mask_request_content or self.mask_response_content:
|
||||
return False
|
||||
|
||||
# if no intervention, return False
|
||||
if response.get("action") != "GUARDRAIL_INTERVENED":
|
||||
return False
|
||||
|
|
|
|||
|
|
@ -166,6 +166,29 @@
|
|||
"litellm_params_template": {
|
||||
"custom_llm_provider": "pydantic_ai_agents"
|
||||
}
|
||||
},
|
||||
{
|
||||
"agent_type": "vertex_agent_engine",
|
||||
"agent_type_display_name": "Vertex AI Agent Engine",
|
||||
"description": "Connect to Google Cloud Vertex AI Reasoning Engines",
|
||||
"logo_url": "/ui/assets/logos/google.svg",
|
||||
"inherit_credentials_from_provider": "Vertex_AI",
|
||||
"model_template": "vertex_ai/agent_engine/{reasoning_engine_id}",
|
||||
"credential_fields": [
|
||||
{
|
||||
"key": "reasoning_engine_id",
|
||||
"label": "Reasoning Engine Resource ID",
|
||||
"placeholder": "projects/123456789/locations/us-central1/reasoningEngines/987654321",
|
||||
"tooltip": "The full resource ID of your Vertex AI Reasoning Engine. Find this in Google Cloud Console under Vertex AI > Agent Builder > Your Agent.",
|
||||
"required": true,
|
||||
"field_type": "text",
|
||||
"default_value": null,
|
||||
"include_in_litellm_params": false
|
||||
}
|
||||
],
|
||||
"litellm_params_template": {
|
||||
"custom_llm_provider": "vertex_ai"
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -2689,8 +2689,8 @@
|
|||
"key": "vertex_credentials",
|
||||
"label": "Vertex Credentials",
|
||||
"placeholder": null,
|
||||
"tooltip": null,
|
||||
"required": true,
|
||||
"tooltip": "Optional - Upload your GCP service account JSON file. If not provided, uses default GCP credentials (ADC).",
|
||||
"required": false,
|
||||
"field_type": "upload",
|
||||
"options": null,
|
||||
"default_value": null
|
||||
|
|
|
|||
|
|
@ -1,3 +1,6 @@
|
|||
# from __future__ import annotations must be the first non-comment statement
|
||||
from __future__ import annotations
|
||||
|
||||
# +-----------------------------------------------+
|
||||
# | |
|
||||
# | Give Feedback / Get Help |
|
||||
|
|
@ -96,11 +99,11 @@ from litellm.litellm_core_utils.core_helpers import (
|
|||
process_response_headers,
|
||||
)
|
||||
from litellm.litellm_core_utils.credential_accessor import CredentialAccessor
|
||||
from litellm.litellm_core_utils.default_encoding import encoding
|
||||
from litellm.litellm_core_utils.dot_notation_indexing import (
|
||||
delete_nested_value,
|
||||
is_nested_path,
|
||||
)
|
||||
from litellm._lazy_imports import _get_default_encoding
|
||||
from litellm.litellm_core_utils.exception_mapping_utils import (
|
||||
_get_response_headers,
|
||||
exception_type,
|
||||
|
|
@ -260,12 +263,16 @@ from litellm.llms.base_llm.base_utils import (
|
|||
BaseLLMModelInfo,
|
||||
type_to_response_format_param,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
# Heavy types that are only needed for type checking; avoid importing
|
||||
# their modules at runtime during `litellm` import.
|
||||
from litellm.llms.base_llm.files.transformation import BaseFilesConfig
|
||||
from litellm.llms.base_llm.batches.transformation import BaseBatchesConfig
|
||||
from litellm.llms.base_llm.chat.transformation import BaseConfig
|
||||
from litellm.llms.base_llm.completion.transformation import BaseTextCompletionConfig
|
||||
from litellm.llms.base_llm.containers.transformation import BaseContainerConfig
|
||||
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
|
||||
from litellm.llms.base_llm.files.transformation import BaseFilesConfig
|
||||
from litellm.llms.base_llm.image_edit.transformation import BaseImageEditConfig
|
||||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
|
|
@ -293,6 +300,7 @@ from .caching.caching import (
|
|||
RedisSemanticCache,
|
||||
S3Cache,
|
||||
)
|
||||
|
||||
from .exceptions import (
|
||||
APIConnectionError,
|
||||
APIError,
|
||||
|
|
@ -1752,7 +1760,7 @@ def _select_tokenizer_helper(model: str) -> SelectTokenizerResponse:
|
|||
|
||||
|
||||
def _return_openai_tokenizer(model: str) -> SelectTokenizerResponse:
|
||||
return {"type": "openai_tokenizer", "tokenizer": encoding}
|
||||
return {"type": "openai_tokenizer", "tokenizer": _get_default_encoding()}
|
||||
|
||||
|
||||
def _return_huggingface_tokenizer(model: str) -> Optional[SelectTokenizerResponse]:
|
||||
|
|
@ -5842,7 +5850,7 @@ def prompt_token_calculator(model, messages):
|
|||
anthropic_obj = Anthropic()
|
||||
num_tokens = anthropic_obj.count_tokens(text) # type: ignore
|
||||
else:
|
||||
num_tokens = len(encoding.encode(text))
|
||||
num_tokens = len(_get_default_encoding().encode(text))
|
||||
return num_tokens
|
||||
|
||||
|
||||
|
|
@ -8187,9 +8195,6 @@ def extract_duration_from_srt_or_vtt(srt_or_vtt_content: str) -> Optional[float]
|
|||
return max(durations) if durations else None
|
||||
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
def _add_path_to_api_base(api_base: str, ending_path: str) -> str:
|
||||
"""
|
||||
Adds an ending path to an API base URL while preventing duplicate path segments.
|
||||
|
|
|
|||
|
|
@ -30628,11 +30628,11 @@
|
|||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "embedding"
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b": {
|
||||
"fireworks_ai/accounts/fireworks/models/": {
|
||||
"max_tokens": 40960,
|
||||
"max_input_tokens": 40960,
|
||||
"max_output_tokens": 40960,
|
||||
"input_cost_per_token": 0.0,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 0.0,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "embedding"
|
||||
|
|
|
|||
|
|
@ -1946,6 +1946,40 @@
|
|||
"rerank": false,
|
||||
"a2a": true
|
||||
}
|
||||
},
|
||||
"vertex_ai/agent_engine": {
|
||||
"display_name": "Vertex AI Agent Engine (`vertex_ai/agent_engine`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/vertex_ai_agent_engine",
|
||||
"endpoints": {
|
||||
"chat_completions": true,
|
||||
"messages": true,
|
||||
"responses": true,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true
|
||||
}
|
||||
},
|
||||
"pydantic_ai_agents": {
|
||||
"display_name": "Pydantic AI Agents (`pydantic_ai_agents`)",
|
||||
"url": "https://docs.litellm.ai/docs/providers/pydantic_ai_agent",
|
||||
"endpoints": {
|
||||
"chat_completions": false,
|
||||
"messages": false,
|
||||
"responses": false,
|
||||
"embeddings": false,
|
||||
"image_generations": false,
|
||||
"audio_transcriptions": false,
|
||||
"audio_speech": false,
|
||||
"moderations": false,
|
||||
"batches": false,
|
||||
"rerank": false,
|
||||
"a2a": true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -152,6 +152,7 @@ model_list:
|
|||
litellm_settings:
|
||||
# set_verbose: True # Uncomment this if you want to see verbose logs; not recommended in production
|
||||
drop_params: True
|
||||
success_callback: ["prometheus"]
|
||||
# max_budget: 100
|
||||
# budget_duration: 30d
|
||||
num_retries: 5
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ uvloop==0.21.0 # uvicorn dep, gives us much better performance under load
|
|||
boto3==1.36.0 # aws bedrock/sagemaker calls
|
||||
redis==5.2.1 # redis caching
|
||||
prisma==0.11.0 # for db
|
||||
nodejs-bin==18.4.0a4 ## required by prisma for migrations, prevents runtime download
|
||||
mangum==0.17.0 # for aws lambda functions
|
||||
pynacl==1.5.0 # for encrypting keys
|
||||
google-cloud-aiplatform==1.47.0 # for vertex ai calls
|
||||
|
|
|
|||
151
tests/agent_tests/local_vertex_agent.py
Normal file
151
tests/agent_tests/local_vertex_agent.py
Normal file
|
|
@ -0,0 +1,151 @@
|
|||
"""
|
||||
Test script for Vertex AI Reasoning Engine.
|
||||
|
||||
This script demonstrates how to:
|
||||
1. Authenticate with Google Cloud
|
||||
2. Send queries to a Vertex AI Reasoning Engine using the :query endpoint
|
||||
|
||||
Usage:
|
||||
python local_vertex_agent.py
|
||||
|
||||
Requirements:
|
||||
pip install httpx google-auth
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
from uuid import uuid4
|
||||
|
||||
from google.auth import default
|
||||
from google.auth.transport.requests import Request
|
||||
import httpx
|
||||
|
||||
# Configuration - update these for your agent
|
||||
PROJECT_ID = "gen-lang-client-0682925754" # Your GCP project ID
|
||||
LOCATION = "us-central1" # Your agent's location
|
||||
|
||||
# For Reasoning Engines, use just the numeric ID at the end
|
||||
REASONING_ENGINE_ID = "8263861224643493888"
|
||||
|
||||
# The project number from the resource name
|
||||
PROJECT_NUMBER = "1060139831167"
|
||||
|
||||
|
||||
async def main():
|
||||
"""Main function to test Vertex AI Reasoning Engine."""
|
||||
|
||||
# Step 1: Authenticate with Google Cloud
|
||||
print("Step 1: Authenticating with Google Cloud...")
|
||||
credentials, project = default(scopes=['https://www.googleapis.com/auth/cloud-platform'])
|
||||
credentials.refresh(Request())
|
||||
print(f"Authenticated! Project: {project}")
|
||||
print(f"Token (first 20 chars): {credentials.token[:20]}...")
|
||||
|
||||
# Step 2: Build the endpoint URL
|
||||
base_url = f"https://{LOCATION}-aiplatform.googleapis.com"
|
||||
resource_path = f"projects/{PROJECT_NUMBER}/locations/{LOCATION}/reasoningEngines/{REASONING_ENGINE_ID}"
|
||||
|
||||
# The Reasoning Engine uses :query endpoint with specific format
|
||||
query_url = f"{base_url}/v1beta1/{resource_path}:query"
|
||||
stream_url = f"{base_url}/v1beta1/{resource_path}:streamQuery"
|
||||
|
||||
print(f"\nQuery URL: {query_url}")
|
||||
print(f"Stream URL: {stream_url}")
|
||||
|
||||
# Step 3: Create authenticated httpx client
|
||||
print("\nStep 2: Creating authenticated HTTP client...")
|
||||
client = httpx.AsyncClient(
|
||||
headers={
|
||||
"Authorization": f"Bearer {credentials.token}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
timeout=120.0,
|
||||
)
|
||||
|
||||
# Step 4: Build the query request (non-streaming)
|
||||
# Note: For non-streaming, we need to:
|
||||
# 1. Create a session
|
||||
# 2. Use the streaming endpoint with stream_query method
|
||||
# The :query endpoint only supports session management methods
|
||||
|
||||
user_id = f"test-user-{uuid4().hex[:8]}"
|
||||
|
||||
# First create a session
|
||||
create_session_request = {
|
||||
"class_method": "async_create_session",
|
||||
"input": {
|
||||
"user_id": user_id,
|
||||
}
|
||||
}
|
||||
|
||||
print(f"\nStep 3: Creating session...")
|
||||
print(f"User ID: {user_id}")
|
||||
|
||||
async with client:
|
||||
# Create session
|
||||
print(f"\nSending to: {query_url}")
|
||||
response = await client.post(query_url, json=create_session_request)
|
||||
print(f"Create session status: {response.status_code}")
|
||||
|
||||
if response.status_code == 200:
|
||||
session_data = response.json()
|
||||
print(f"Session created:\n{json.dumps(session_data, indent=2)}")
|
||||
|
||||
# Extract session_id from response
|
||||
session_id = session_data.get("output", {}).get("id") or session_data.get("output", {}).get("session_id")
|
||||
print(f"\nSession ID: {session_id}")
|
||||
|
||||
# Now send the actual query via streamQuery
|
||||
query_request = {
|
||||
"class_method": "stream_query",
|
||||
"input": {
|
||||
"message": "Hello! What can you do?",
|
||||
"user_id": user_id,
|
||||
"session_id": session_id,
|
||||
}
|
||||
}
|
||||
|
||||
print(f"\nStep 4: Sending query via streamQuery...")
|
||||
print(f"Request:\n{json.dumps(query_request, indent=2)}")
|
||||
|
||||
# Use streaming endpoint but collect full response
|
||||
async with client.stream("POST", stream_url, json=query_request) as stream_response:
|
||||
print(f"Query status: {stream_response.status_code}")
|
||||
|
||||
if stream_response.status_code == 200:
|
||||
print("\nResponse:")
|
||||
full_response = ""
|
||||
async for line in stream_response.aiter_lines():
|
||||
if line:
|
||||
full_response = line # Keep last line (full response)
|
||||
|
||||
# Parse and display
|
||||
try:
|
||||
data = json.loads(full_response)
|
||||
# Extract the text from the response
|
||||
content = data.get("content", {})
|
||||
parts = content.get("parts", [])
|
||||
for part in parts:
|
||||
if "text" in part:
|
||||
print(f"\nAgent response:\n{part['text']}")
|
||||
except:
|
||||
print(full_response)
|
||||
else:
|
||||
content = await stream_response.aread()
|
||||
print(f"Error: {content.decode()}")
|
||||
else:
|
||||
print(f"Error creating session: {response.text}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("=" * 60)
|
||||
print("Vertex AI Reasoning Engine Test Script")
|
||||
print("=" * 60)
|
||||
print(f"\nConfiguration:")
|
||||
print(f" PROJECT_ID: {PROJECT_ID}")
|
||||
print(f" PROJECT_NUMBER: {PROJECT_NUMBER}")
|
||||
print(f" LOCATION: {LOCATION}")
|
||||
print(f" REASONING_ENGINE_ID: {REASONING_ENGINE_ID}")
|
||||
print()
|
||||
|
||||
asyncio.run(main())
|
||||
|
|
@ -201,3 +201,79 @@ async def test_a2a_completion_bridge_bedrock_agentcore():
|
|||
|
||||
print(f"Received {len(chunks)} chunks from Bedrock AgentCore")
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Vertex AI Agent Engine Tests
|
||||
# ============================================================
|
||||
|
||||
# Configuration - update these for your Vertex AI Reasoning Engine
|
||||
VERTEX_AGENT_RESOURCE_NAME = "projects/1060139831167/locations/us-central1/reasoningEngines/8263861224643493888"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_vertex_agent_engine_non_streaming():
|
||||
"""
|
||||
Test non-streaming request to Vertex AI Agent Engine via litellm.acompletion.
|
||||
|
||||
Uses the Reasoning Engine resource ID to call a hosted agent.
|
||||
"""
|
||||
|
||||
litellm._turn_on_debug()
|
||||
|
||||
# Call via litellm.acompletion with vertex_ai/agent_engine/ prefix
|
||||
response = await litellm.acompletion(
|
||||
model=f"vertex_ai/agent_engine/{VERTEX_AGENT_RESOURCE_NAME}",
|
||||
messages=[{"role": "user", "content": "Hello! What can you do?"}],
|
||||
stream=False,
|
||||
)
|
||||
|
||||
print(f"\n=== Vertex Agent Engine Non-Streaming Response ===")
|
||||
print(f"Response: {response}")
|
||||
|
||||
# Basic assertions
|
||||
assert response is not None
|
||||
assert hasattr(response, "choices")
|
||||
assert len(response.choices) > 0
|
||||
assert response.choices[0].message is not None
|
||||
assert response.choices[0].message.content is not None
|
||||
assert len(response.choices[0].message.content) > 0
|
||||
|
||||
print(f"Agent response: {response.choices[0].message.content[:200]}...")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_vertex_agent_engine_streaming():
|
||||
"""
|
||||
Test streaming request to Vertex AI Agent Engine via litellm.acompletion.
|
||||
|
||||
Uses the Reasoning Engine resource ID to call a hosted agent with streaming.
|
||||
"""
|
||||
#litellm._turn_on_debug()
|
||||
|
||||
# Call via litellm.acompletion with streaming
|
||||
response = await litellm.acompletion(
|
||||
model=f"vertex_ai/agent_engine/{VERTEX_AGENT_RESOURCE_NAME}",
|
||||
messages=[{"role": "user", "content": "Hello! What can you do?"}],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
print(f"\n=== Vertex Agent Engine Streaming Response ===")
|
||||
|
||||
chunks = []
|
||||
full_content = ""
|
||||
async for chunk in response:
|
||||
print(f"Chunk: {chunk}")
|
||||
# chunks.append(chunk)
|
||||
# if hasattr(chunk, "choices") and len(chunk.choices) > 0:
|
||||
# delta = chunk.choices[0].delta
|
||||
# if hasattr(delta, "content") and delta.content:
|
||||
# full_content += delta.content
|
||||
# print(f"Chunk: {delta.content}", end="", flush=True)
|
||||
|
||||
# # print(f"\n\nReceived {len(chunks)} chunks")
|
||||
# print(f"Full content: {full_content[:200]}...")
|
||||
|
||||
# # Basic assertions
|
||||
# assert len(chunks) > 0
|
||||
# assert len(full_content) > 0
|
||||
|
||||
|
|
|
|||
128
tests/litellm/llms/vertex_ai/agent_engine/test_transformation.py
Normal file
128
tests/litellm/llms/vertex_ai/agent_engine/test_transformation.py
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
"""
|
||||
Tests for Vertex AI Agent Engine transformation.
|
||||
|
||||
Tests the request transformation and streaming chunk parsing without making real API calls.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
|
||||
from litellm.llms.vertex_ai.agent_engine.sse_iterator import (
|
||||
VertexAgentEngineResponseIterator,
|
||||
)
|
||||
from litellm.llms.vertex_ai.agent_engine.transformation import VertexAgentEngineConfig
|
||||
|
||||
|
||||
class TestVertexAgentEngineTransformRequest:
|
||||
"""Tests for transform_request method."""
|
||||
|
||||
def test_transform_request_basic(self):
|
||||
"""
|
||||
Test that transform_request correctly formats messages into Vertex Agent Engine payload.
|
||||
"""
|
||||
config = VertexAgentEngineConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Hello, what can you do?"}]
|
||||
optional_params = {"user_id": "test-user-123"}
|
||||
litellm_params = {}
|
||||
|
||||
result = config.transform_request(
|
||||
model="agent_engine/123456789",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["class_method"] == "stream_query"
|
||||
assert result["input"]["message"] == "Hello, what can you do?"
|
||||
assert result["input"]["user_id"] == "test-user-123"
|
||||
assert "session_id" not in result["input"]
|
||||
|
||||
def test_transform_request_with_session_id(self):
|
||||
"""
|
||||
Test that transform_request includes session_id when provided.
|
||||
"""
|
||||
config = VertexAgentEngineConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Follow up question"}]
|
||||
optional_params = {
|
||||
"user_id": "test-user-123",
|
||||
"session_id": "session-abc-456",
|
||||
}
|
||||
litellm_params = {}
|
||||
|
||||
result = config.transform_request(
|
||||
model="agent_engine/123456789",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["class_method"] == "stream_query"
|
||||
assert result["input"]["message"] == "Follow up question"
|
||||
assert result["input"]["user_id"] == "test-user-123"
|
||||
assert result["input"]["session_id"] == "session-abc-456"
|
||||
|
||||
|
||||
class TestVertexAgentEngineChunkParser:
|
||||
"""Tests for the streaming chunk parser."""
|
||||
|
||||
def test_chunk_parser_with_text_content(self):
|
||||
"""
|
||||
Test that chunk_parser correctly extracts text from Vertex Agent Engine response format.
|
||||
"""
|
||||
iterator = VertexAgentEngineResponseIterator(
|
||||
streaming_response=iter([]),
|
||||
sync_stream=True,
|
||||
)
|
||||
|
||||
chunk = {
|
||||
"content": {
|
||||
"parts": [{"text": "Hello! I can help you with financial analysis."}],
|
||||
"role": "model",
|
||||
},
|
||||
"finish_reason": "STOP",
|
||||
"usage_metadata": {
|
||||
"prompt_token_count": 100,
|
||||
"candidates_token_count": 50,
|
||||
"total_token_count": 150,
|
||||
},
|
||||
}
|
||||
|
||||
result = iterator.chunk_parser(chunk)
|
||||
|
||||
assert result.choices[0].delta.content == "Hello! I can help you with financial analysis."
|
||||
assert result.choices[0].delta.role == "assistant"
|
||||
assert result.choices[0].finish_reason == "stop"
|
||||
assert result.usage["prompt_tokens"] == 100
|
||||
assert result.usage["completion_tokens"] == 50
|
||||
assert result.usage["total_tokens"] == 150
|
||||
|
||||
def test_chunk_parser_without_finish_reason(self):
|
||||
"""
|
||||
Test that chunk_parser handles chunks without finish_reason (intermediate chunks).
|
||||
"""
|
||||
iterator = VertexAgentEngineResponseIterator(
|
||||
streaming_response=iter([]),
|
||||
sync_stream=True,
|
||||
)
|
||||
|
||||
chunk = {
|
||||
"content": {
|
||||
"parts": [{"text": "Partial response..."}],
|
||||
"role": "model",
|
||||
},
|
||||
}
|
||||
|
||||
result = iterator.chunk_parser(chunk)
|
||||
|
||||
assert result.choices[0].delta.content == "Partial response..."
|
||||
assert result.choices[0].finish_reason is None
|
||||
assert result.usage is None
|
||||
|
||||
|
|
@ -182,3 +182,94 @@ async def test_azure_responses_api_status_error():
|
|||
f"Expected: {json.dumps(expected_input, indent=2)}\n"
|
||||
f"Got: {json.dumps(captured_request_body['input'], indent=2)}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_azure_responses_api_headers_with_llm_provider_prefix():
|
||||
"""
|
||||
Test that Azure-specific headers like 'x-request-id' and 'apim-request-id'
|
||||
are properly forwarded with 'llm_provider-' prefix in response._hidden_params["headers"].
|
||||
|
||||
Issue: https://github.com/BerriAI/litellm/issues/16538
|
||||
|
||||
The fix ensures that processed headers (with llm_provider- prefix) are stored
|
||||
in response._hidden_params["headers"] instead of additional_headers, making them
|
||||
accessible via completion.headers in the same way as the completion API.
|
||||
"""
|
||||
import json
|
||||
import httpx
|
||||
|
||||
mock_response_data = {
|
||||
"id": "resp_123",
|
||||
"object": "response",
|
||||
"created_at": 1234567890,
|
||||
"model": "gpt-5-codex",
|
||||
"status": "completed",
|
||||
"output": [
|
||||
{
|
||||
"id": "msg_123",
|
||||
"role": "assistant",
|
||||
"type": "message",
|
||||
"content": [{"type": "output_text", "text": "Hello!"}],
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
# Mock headers that Azure returns - exactly like in the issue
|
||||
mock_headers = {
|
||||
"date": "Wed, 12 Nov 2025 15:31:28 GMT",
|
||||
"server": "uvicorn",
|
||||
"content-type": "application/json",
|
||||
"x-ratelimit-remaining-tokens": "5010000",
|
||||
"x-ratelimit-limit-tokens": "5010000",
|
||||
# These are the Azure-specific headers that should be forwarded with llm_provider- prefix
|
||||
"x-request-id": "12086715-aca3-4006-a29f-2f1e1d552043",
|
||||
"apim-request-id": "25664b0d-cf4b-4e10-8d27-c7272e7efd49",
|
||||
"x-ms-region": "Sweden Central",
|
||||
}
|
||||
|
||||
async def mock_post(*args, **kwargs):
|
||||
response_content = json.dumps(mock_response_data).encode("utf-8")
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
headers=mock_headers,
|
||||
content=response_content,
|
||||
request=httpx.Request(method="POST", url="https://test.openai.azure.com"),
|
||||
)
|
||||
return response
|
||||
|
||||
with patch.object(AsyncHTTPHandler, "post", new=mock_post):
|
||||
response = await litellm.aresponses(
|
||||
model="azure/gpt-5-codex",
|
||||
api_version="2025-03-01-preview",
|
||||
api_base="https://test.openai.azure.com",
|
||||
api_key="test-key",
|
||||
input="Hello, can you tell me a short joke?",
|
||||
)
|
||||
|
||||
# Check that the response has the expected headers structure
|
||||
assert hasattr(response, "_hidden_params"), "Response should have _hidden_params"
|
||||
assert "additional_headers" in response._hidden_params, (
|
||||
"Response _hidden_params should contain 'additional_headers' with the LLM provider headers"
|
||||
)
|
||||
|
||||
headers = response._hidden_params["additional_headers"]
|
||||
|
||||
# Verify that Azure-specific headers are present with llm_provider- prefix
|
||||
assert "llm_provider-x-request-id" in headers, (
|
||||
f"Response should contain 'llm_provider-x-request-id' header. "
|
||||
f"Headers: {list(headers.keys())}"
|
||||
)
|
||||
assert "llm_provider-apim-request-id" in headers, (
|
||||
f"Response should contain 'llm_provider-apim-request-id' header. "
|
||||
f"Headers: {list(headers.keys())}"
|
||||
)
|
||||
|
||||
# Verify the header values match
|
||||
assert headers["llm_provider-x-request-id"] == "12086715-aca3-4006-a29f-2f1e1d552043"
|
||||
assert headers["llm_provider-apim-request-id"] == "25664b0d-cf4b-4e10-8d27-c7272e7efd49"
|
||||
assert headers["llm_provider-x-ms-region"] == "Sweden Central"
|
||||
|
||||
# Also verify openai-compatible headers are included
|
||||
assert "x-ratelimit-limit-tokens" in headers
|
||||
assert "x-ratelimit-remaining-tokens" in headers
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue