Merge remote-tracking branch 'upstream/main' into litellm_fix_langfuse_init

This commit is contained in:
Thiago Riemma Carbonera 2026-04-01 15:28:21 -03:00
commit ade7a64ce4
349 changed files with 15348 additions and 8865 deletions

File diff suppressed because it is too large Load diff

View file

@ -12,6 +12,6 @@ echo "[post-create] Generating Prisma client"
poetry run prisma generate
echo "[post-create] Installing npm dependencies"
cd ui/litellm-dashboard && npm install --no-audit --no-fund
cd ui/litellm-dashboard && npm ci
echo "[post-create] Done"

View file

@ -41,32 +41,54 @@ runs:
using: composite
steps:
- name: Helm | Setup
uses: azure/setup-helm@v4
uses: azure/setup-helm@1a275c3b69536ee54be43f2070a358922e12c8d4 # v4.3.1
with:
version: v3.20.0
- name: Helm | Login
shell: bash
run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }}
env:
REGISTRY_PASSWORD: ${{ inputs.registry_password }}
REGISTRY_USERNAME: ${{ inputs.registry_username }}
REGISTRY: ${{ inputs.registry }}
run: echo "$REGISTRY_PASSWORD" | helm registry login -u "$REGISTRY_USERNAME" --password-stdin "$REGISTRY"
- name: Helm | Dependency
if: inputs.update_dependencies == 'true'
shell: bash
run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
env:
CHART_PATH: ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
run: helm dependency update "$CHART_PATH"
- name: Helm | Package
shell: bash
run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }}
env:
CHART_PATH: ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
TAG: ${{ inputs.tag }}
APP_VERSION: ${{ inputs.app_version }}
run: helm package "$CHART_PATH" --version "$TAG" --app-version "$APP_VERSION"
- name: Helm | Push
shell: bash
run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }}
env:
NAME: ${{ inputs.name }}
TAG: ${{ inputs.tag }}
REGISTRY: ${{ inputs.registry }}
REPOSITORY: ${{ inputs.repository }}
run: helm push "${NAME}-${TAG}.tgz" "oci://${REGISTRY}/${REPOSITORY}"
- name: Helm | Logout
shell: bash
run: helm registry logout ${{ inputs.registry }}
env:
REGISTRY: ${{ inputs.registry }}
run: helm registry logout "$REGISTRY"
- name: Helm | Output
id: output
shell: bash
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
env:
REGISTRY: ${{ inputs.registry }}
REPOSITORY: ${{ inputs.repository }}
NAME: ${{ inputs.name }}
TAG: ${{ inputs.tag }}
run: echo "image=${REGISTRY}/${REPOSITORY}/${NAME}:${TAG}" >> $GITHUB_OUTPUT

View file

@ -1,22 +1,21 @@
name: "LiteLLM CodeQL config"
# Use security-extended suite instead of security-and-quality to avoid
# result sets > 2 GiB on this codebase that cause fatal OOM failures.
queries:
- uses: security-extended
- uses: security-and-quality
# These two queries are security queries included in security-extended that
# individually produce result sets > 2 GiB on this codebase, causing fatal
# OOM failures. Exclude them as a safety net until CI confirms they no longer
# OOM; drop these exclusions in a follow-up once verified.
# Known OOM queries on large Python codebases:
# CodeQL builds a full data flow graph in memory. These two queries trace
# sensitive data through every log call / regex pattern, causing combinatorial
# path explosion on codebases with extensive logging like LiteLLM (>2 GiB
# result sets). This is a known CodeQL scaling limitation, not a code issue.
# Re-test periodically as CodeQL improves or the codebase refactors logging.
query-filters:
- exclude:
id: py/clear-text-logging-sensitive-data # CWE-312 — > 2 GiB result set
id: py/clear-text-logging-sensitive-data # CWE-312
- exclude:
id: py/polynomial-redos # CWE-730 — > 2 GiB result set
id: py/polynomial-redos # CWE-730
paths-ignore:
- tests
- docs
- "**/*.md"
- litellm/proxy/_experimental/out

View file

@ -4,6 +4,9 @@ updates:
directory: "/"
schedule:
interval: "daily"
cooldown:
default-days: 7
semver-major-days: 14
groups:
github-actions:
patterns:

96
.github/workflows/_test-unit-base.yml vendored Normal file
View file

@ -0,0 +1,96 @@
name: _Unit Test Base (Reusable)
on:
workflow_call:
inputs:
test-path:
description: "Pytest path(s) to run"
required: true
type: string
workers:
description: "Number of pytest-xdist workers"
required: false
type: number
default: 2
reruns:
description: "Number of reruns for flaky tests"
required: false
type: number
default: 2
timeout-minutes:
description: "Job timeout in minutes"
required: false
type: number
default: 20
max-failures:
description: "Stop after this many failures"
required: false
type: number
default: 10
permissions:
contents: read
jobs:
run:
name: Run tests
runs-on: ubuntu-latest
timeout-minutes: ${{ inputs.timeout-minutes }}
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Install Poetry
run: pip install 'poetry==2.3.2'
- name: Cache Poetry dependencies
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/pypoetry
~/.cache/pip
.venv
key: ${{ runner.os }}-poetry-${{ hashFiles('poetry.lock') }}
restore-keys: |
${{ runner.os }}-poetry-
- name: Install dependencies
run: |
poetry config virtualenvs.in-project true
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install google-genai==1.22.0 \
google-cloud-aiplatform==1.115.0 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core==0.23.0
- name: Setup litellm-enterprise
run: |
poetry run pip install --force-reinstall --no-deps -e enterprise/
- name: Generate Prisma client
env:
PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache
run: |
poetry run pip install nodejs-wheel-binaries==24.13.1
poetry run prisma generate --schema litellm/proxy/schema.prisma
- name: Run tests
env:
TEST_PATH: ${{ inputs.test-path }}
MAX_FAILURES: ${{ inputs.max-failures }}
WORKERS: ${{ inputs.workers }}
RERUNS: ${{ inputs.reruns }}
run: |
poetry run pytest ${TEST_PATH:?} \
--tb=short -vv \
--maxfail="${MAX_FAILURES}" \
-n "${WORKERS}" \
--reruns "${RERUNS}" \
--reruns-delay 1 \
--dist=loadscope \
--durations=20

View file

@ -0,0 +1,164 @@
name: _Unit Test Services Base (Reusable)
on:
workflow_call:
inputs:
test-path:
description: "Pytest path(s) to run"
required: true
type: string
workers:
description: "Number of pytest-xdist workers (0 = no parallelism)"
required: false
type: number
default: 2
reruns:
description: "Number of reruns for flaky tests"
required: false
type: number
default: 2
timeout-minutes:
description: "Job timeout in minutes"
required: false
type: number
default: 20
max-failures:
description: "Stop after this many failures"
required: false
type: number
default: 10
enable-redis:
description: "Pass Redis Cloud credentials to tests via REDIS_HOST/PORT/PASSWORD env vars"
required: false
type: boolean
default: false
enable-postgres:
description: "Start a local Postgres service container and run Prisma migrations"
required: false
type: boolean
default: false
secrets:
REDIS_HOST:
required: false
REDIS_PORT:
required: false
REDIS_PASSWORD:
required: false
DATABASE_URL:
required: false
POSTGRES_USER:
required: false
POSTGRES_PASSWORD:
required: false
permissions:
contents: read
jobs:
run:
name: Run tests
runs-on: ubuntu-latest
timeout-minutes: ${{ inputs.timeout-minutes }}
# Environment is derived from the enable-* flags, not caller-controllable.
# This prevents callers from passing arbitrary environment names to bypass secret scoping.
# Note: Postgres service container always starts (GHA limitation), so any Redis job
# also needs Postgres secrets → uses integration-redis-postgres, not integration-redis.
environment: >-
${{
inputs.enable-redis && 'integration-redis-postgres' ||
inputs.enable-postgres && 'integration-postgres' ||
''
}}
services:
postgres:
image: postgres@sha256:705a5d5b5836f3fcba0d02c4d281e6a7dd9ed2dd4078640f08a1e1e9896e097d # postgres:14
env:
POSTGRES_USER: ${{ secrets.POSTGRES_USER }}
POSTGRES_PASSWORD: ${{ secrets.POSTGRES_PASSWORD }}
POSTGRES_DB: litellm_test
ports:
- 5432:5432
options: >-
--health-cmd "pg_isready"
--health-interval 10s
--health-timeout 5s
--health-retries 5
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Install Poetry
run: pip install 'poetry==2.3.2'
- name: Cache Poetry dependencies
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/pypoetry
~/.cache/pip
.venv
key: ${{ runner.os }}-poetry-services-${{ hashFiles('poetry.lock') }}
restore-keys: |
${{ runner.os }}-poetry-services-
- name: Install dependencies
run: |
poetry config virtualenvs.in-project true
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install google-genai==1.22.0 \
google-cloud-aiplatform==1.115.0 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core==0.23.0
- name: Setup litellm-enterprise
run: |
poetry run pip install --force-reinstall --no-deps -e enterprise/
- name: Generate Prisma client
env:
PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache
run: |
poetry run pip install nodejs-wheel-binaries==24.13.1
poetry run prisma generate --schema litellm/proxy/schema.prisma
- name: Run Prisma migrations
if: ${{ inputs.enable-postgres }}
env:
DATABASE_URL: ${{ secrets.DATABASE_URL }}
run: |
poetry run prisma db push --schema litellm/proxy/schema.prisma --accept-data-loss
- name: Run tests
env:
TEST_PATH: ${{ inputs.test-path }}
MAX_FAILURES: ${{ inputs.max-failures }}
WORKERS: ${{ inputs.workers }}
RERUNS: ${{ inputs.reruns }}
DATABASE_URL: ${{ inputs.enable-postgres && secrets.DATABASE_URL || '' }}
REDIS_HOST: ${{ inputs.enable-redis && secrets.REDIS_HOST || '' }}
REDIS_PORT: ${{ inputs.enable-redis && secrets.REDIS_PORT || '' }}
REDIS_PASSWORD: ${{ inputs.enable-redis && secrets.REDIS_PASSWORD || '' }}
run: |
if [ "${WORKERS}" = "0" ]; then
poetry run pytest ${TEST_PATH:?} \
--tb=short -vv \
--maxfail="${MAX_FAILURES}" \
--reruns "${RERUNS}" \
--reruns-delay 1 \
--durations=20
else
poetry run pytest ${TEST_PATH:?} \
--tb=short -vv \
--maxfail="${MAX_FAILURES}" \
-n "${WORKERS}" \
--reruns "${RERUNS}" \
--reruns-delay 1 \
--dist=loadscope \
--durations=20
fi

58
.github/workflows/check-schema-sync.yml vendored Normal file
View file

@ -0,0 +1,58 @@
name: Check Schema Sync
on:
pull_request:
paths:
- 'schema.prisma'
- 'litellm/proxy/schema.prisma'
- 'litellm-proxy-extras/litellm_proxy_extras/schema.prisma'
permissions:
contents: read
jobs:
check-sync:
name: Verify schema.prisma copies match root
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Checkout PR
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Reject symlinked schema files
run: |
for f in schema.prisma litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; do
if [ -L "$f" ]; then
echo "::error file=$f::$f is a symlink, which is not allowed"
exit 1
fi
done
- name: Check all schemas match root
run: |
EXIT=0
diff schema.prisma litellm/proxy/schema.prisma || {
echo "::error file=litellm/proxy/schema.prisma::litellm/proxy/schema.prisma differs from root schema.prisma"
EXIT=1
}
diff schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma || {
echo "::error file=litellm-proxy-extras/litellm_proxy_extras/schema.prisma::litellm-proxy-extras/litellm_proxy_extras/schema.prisma differs from root schema.prisma"
EXIT=1
}
if [ "$EXIT" -ne 0 ]; then
echo ""
echo "Schema files are out of sync."
echo "The root schema.prisma is the source of truth."
echo ""
echo "To fix, run from the repo root:"
echo " cp schema.prisma litellm/proxy/schema.prisma"
echo " cp schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma"
exit 1
fi
echo "All schema copies are in sync with root."

93
.github/workflows/create-release.yml vendored Normal file
View file

@ -0,0 +1,93 @@
name: Create Release
on:
workflow_dispatch:
inputs:
tag:
description: "Release tag (e.g. v1.83.0-stable)"
required: true
type: string
commit_hash:
description: "Full 40-char commit SHA to target"
required: true
type: string
permissions: {}
jobs:
release:
name: Create Release
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- name: Validate inputs
env:
TAG: ${{ inputs.tag }}
COMMIT_HASH: ${{ inputs.commit_hash }}
run: |
if ! echo "${COMMIT_HASH}" | grep -qE '^[0-9a-f]{40}$'; then
echo "::error::commit_hash must be a full 40-character commit SHA"
exit 1
fi
if ! echo "${TAG}" | grep -qE '^v[0-9]+\.[0-9]+\.[0-9]+'; then
echo "::error::tag must start with vX.Y.Z"
exit 1
fi
- name: Create release
env:
TAG: ${{ inputs.tag }}
COMMIT_HASH: ${{ inputs.commit_hash }}
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const tag = process.env.TAG;
const commitHash = process.env.COMMIT_HASH;
const cosignSection = [
`## Verify Docker Image Signature`,
``,
`All LiteLLM Docker images are signed with [cosign](https://docs.sigstore.dev/cosign/overview/). To verify the integrity of an image before deploying:`,
``,
'```bash',
`cosign verify \\`,
` --key https://raw.githubusercontent.com/BerriAI/litellm/${tag}/cosign.pub \\`,
` ghcr.io/berriai/litellm:${tag}`,
'```',
``,
`Expected output:`,
``,
'```',
`The following checks were performed on each of these signatures:`,
` - The cosign claims were validated`,
` - The signatures were verified against the specified public key`,
'```',
``,
`---`,
``,
].join('\n');
try {
const response = await github.rest.repos.createRelease({
draft: true,
generate_release_notes: true,
target_commitish: commitHash,
name: tag,
owner: context.repo.owner,
prerelease: false,
repo: context.repo.repo,
tag_name: tag,
});
const updatedBody = cosignSection + (response.data.body ?? '');
await github.rest.repos.updateRelease({
owner: context.repo.owner,
repo: context.repo.repo,
release_id: response.data.id,
body: updatedBody,
draft: false,
});
} catch (error) {
core.setFailed(error.message);
}

136
.github/workflows/publish_to_pypi.yml vendored Normal file
View file

@ -0,0 +1,136 @@
name: Publish to PyPI
on:
workflow_dispatch:
jobs:
preflight-checks:
name: Preflight Checks
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
# No environment — read-only checks, no approval needed
outputs:
needs_publish: ${{ steps.check-litellm.outputs.needs_publish }}
version: ${{ steps.check-litellm.outputs.version }}
steps:
- name: Checkout repo
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Check litellm version on PyPI
id: check-litellm
run: |
VERSION=$(grep -m1 '^version' pyproject.toml | sed 's/version = "\(.*\)"/\1/')
echo "version=$VERSION" >> "$GITHUB_OUTPUT"
echo "Checking if litellm $VERSION exists on PyPI..."
HTTP_STATUS=$(curl -s -o /dev/null -w "%{http_code}" "https://pypi.org/pypi/litellm/$VERSION/json")
if [ "$HTTP_STATUS" = "200" ]; then
echo "litellm $VERSION already exists on PyPI. Skipping publish."
echo "needs_publish=false" >> "$GITHUB_OUTPUT"
else
echo "litellm $VERSION not found on PyPI. Publish needed."
echo "needs_publish=true" >> "$GITHUB_OUTPUT"
fi
- name: Sanity check proxy-extras version
run: |
# Read pinned version from requirements.txt
REQ_VERSION=$(grep -oP 'litellm-proxy-extras==\K[0-9.]+' requirements.txt)
if [ -z "$REQ_VERSION" ]; then
echo "::error::Could not find litellm-proxy-extras version in requirements.txt"
exit 1
fi
echo "requirements.txt pins litellm-proxy-extras==$REQ_VERSION"
# Read pinned version from pyproject.toml dependency
PYPROJECT_VERSION=$(python3 -c "
import re
with open('pyproject.toml') as f:
content = f.read()
match = re.search(r'litellm-proxy-extras\s*=\s*\{version\s*=\s*\"([^\"]+)\"', content)
if match:
print(match.group(1).lstrip('^~>='))
else:
import sys
print('::error::Could not find litellm-proxy-extras dependency in pyproject.toml', file=sys.stderr)
sys.exit(1)
")
echo "pyproject.toml pins litellm-proxy-extras version: $PYPROJECT_VERSION"
# Check that both pinned versions match
if [ "$REQ_VERSION" != "$PYPROJECT_VERSION" ]; then
echo "::error::Version mismatch: requirements.txt has $REQ_VERSION but pyproject.toml has $PYPROJECT_VERSION"
exit 1
fi
# Check that the pinned version exists on PyPI
echo "Checking if litellm-proxy-extras $REQ_VERSION exists on PyPI..."
HTTP_STATUS=$(curl -s -o /dev/null -w "%{http_code}" "https://pypi.org/pypi/litellm-proxy-extras/$REQ_VERSION/json")
if [ "$HTTP_STATUS" != "200" ]; then
echo "::error::litellm-proxy-extras $REQ_VERSION is not published on PyPI yet. Publish it before releasing litellm."
exit 1
fi
echo "litellm-proxy-extras $REQ_VERSION exists on PyPI. Sanity check passed."
publish-litellm:
name: Publish litellm to PyPI
needs: preflight-checks
if: needs.preflight-checks.outputs.needs_publish == 'true'
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
id-token: write
contents: read
environment: pypi-publish
steps:
- name: Checkout repo
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Copy model prices backup
run: cp model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json
- name: Install build tools
run: python -m pip install --upgrade pip build==1.4.2
- name: Build package
run: |
rm -rf build dist
python -m build
- name: Verify build artifacts
env:
EXPECTED_VERSION: ${{ needs.preflight-checks.outputs.version }}
run: |
echo "Contents of dist/:"
ls -la dist/
# Ensure we have both sdist and wheel
ls dist/*.tar.gz
ls dist/*.whl
# Verify built version matches expected
ls dist/ | grep -q "litellm-${EXPECTED_VERSION}" || {
echo "::error::Built artifacts do not match expected version $EXPECTED_VERSION"
ls dist/
exit 1
}
- name: Validate package metadata
run: |
pip install twine==6.2.0
twine check dist/*
- name: Publish to PyPI
uses: pypa/gh-action-pypi-publish@ed0c53931b1dc9bd32cbe73a98c7f6766f8a527e # v1.13.0

47
.github/workflows/scorecard.yml vendored Normal file
View file

@ -0,0 +1,47 @@
name: Scorecard supply-chain security
on:
branch_protection_rule:
schedule:
- cron: '27 12 * * 4'
push:
branches: ["main"]
permissions: read-all
jobs:
analysis:
name: Scorecard analysis
runs-on: ubuntu-latest
if: github.event.repository.default_branch == github.ref_name
permissions:
security-events: write
id-token: write
# Uncomment for private repos if needed:
# contents: read
# actions: read
steps:
- name: Checkout code
uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
with:
persist-credentials: false
- name: Run analysis
uses: ossf/scorecard-action@f49aabe0b5af0936a0987cfb85d86b75731b0186 # v2.4.1
with:
results_file: results.sarif
results_format: sarif
publish_results: true
- name: Upload artifact
uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1
with:
name: SARIF file
path: results.sarif
retention-days: 5
- name: Upload to code scanning
uses: github/codeql-action/upload-sarif@c10b806170c8ee63ea24152429041b5624f0baf5 # v4.35.1
with:
sarif_file: results.sarif

73
.github/workflows/sync-schema.yml vendored Normal file
View file

@ -0,0 +1,73 @@
name: Sync schema.prisma copies
on:
pull_request:
paths:
- 'schema.prisma'
# Scoped to ONLY the permissions needed:
# - contents:write to push the sync commit to the PR branch
# - pull-requests:read is implicit (needed to check out the PR)
permissions:
contents: write
jobs:
sync:
name: Copy root schema to proxy and proxy-extras
runs-on: ubuntu-latest
timeout-minutes: 5
# Only run on PRs from branches in THIS repo (not forks).
# Fork PRs cannot push back to the head branch with GITHUB_TOKEN,
# and pull_request events from forks have read-only tokens anyway.
# Also reject PRs from branches named after protected branches to
# prevent pushing directly to main/master.
if: >-
github.event.pull_request.head.repo.full_name == github.repository
&& github.head_ref != 'main'
&& github.head_ref != 'master'
steps:
- name: Checkout PR branch by SHA
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
# Use the merge commit SHA for safety — github.head_ref is an
# attacker-controlled string (the branch name) and could contain
# unusual characters that cause unexpected git behavior.
ref: ${{ github.event.pull_request.head.sha }}
persist-credentials: true # needed for git push
- name: Reject symlinked schema files
run: |
for f in schema.prisma litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; do
if [ -L "$f" ]; then
echo "::error file=$f::$f is a symlink, which is not allowed"
exit 1
fi
done
- name: Copy root schema to other locations
run: |
cp schema.prisma litellm/proxy/schema.prisma
cp schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma
- name: Check for changes
id: diff
run: |
if git diff --quiet -- litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; then
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "Schemas already in sync. Nothing to do."
else
echo "changed=true" >> "$GITHUB_OUTPUT"
echo "Schema copies need updating."
fi
- name: Commit synced schemas
if: steps.diff.outputs.changed == 'true'
run: |
# Push to the PR's head branch (need the branch name for git push).
# We checked out by SHA above for safety, so configure the push target explicitly.
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git checkout -B "$GITHUB_HEAD_REF"
git add -- litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma
git commit -m "chore: sync schema.prisma copies from root"
git push origin "HEAD:$GITHUB_HEAD_REF"

View file

@ -0,0 +1,38 @@
name: "Unit Tests: Caching (Redis)"
# Uses cloud Redis credentials — only runs on trusted branches, not PRs.
# This prevents external PRs from accessing Redis credentials.
on:
push:
branches: [main, "litellm_*"]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
caching-redis:
uses: ./.github/workflows/_test-unit-services-base.yml
with:
# Redis-only tests that do NOT require provider API keys.
# Tests needing API keys (test_caching.py, test_caching_ssl.py, test_prometheus_service.py,
# test_router_caching.py) are in Phase 3 integration workflows.
test-path: >-
tests/local_testing/test_dual_cache.py
tests/local_testing/test_redis_batch_optimizations.py
tests/local_testing/test_router_utils.py
workers: 2
reruns: 2
timeout-minutes: 20
enable-redis: true
enable-postgres: false
secrets:
REDIS_HOST: ${{ secrets.REDIS_HOST }}
REDIS_PORT: ${{ secrets.REDIS_PORT }}
REDIS_PASSWORD: ${{ secrets.REDIS_PASSWORD }}
DATABASE_URL: ${{ secrets.DATABASE_URL }}
POSTGRES_USER: ${{ secrets.POSTGRES_USER }}
POSTGRES_PASSWORD: ${{ secrets.POSTGRES_PASSWORD }}

View file

@ -0,0 +1,20 @@
name: "Unit Tests: Core Utilities"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
core-utils:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/litellm_core_utils"
workers: 2
reruns: 1

View file

@ -0,0 +1,67 @@
name: "Unit Tests: Documentation Validation"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
documentation:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Install Poetry
run: pip install 'poetry==2.3.2'
- name: Cache Poetry dependencies
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/pypoetry
~/.cache/pip
.venv
key: ${{ runner.os }}-poetry-${{ hashFiles('poetry.lock') }}
restore-keys: |
${{ runner.os }}-poetry-
- name: Install dependencies
run: |
poetry config virtualenvs.in-project true
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install google-genai==1.22.0 \
google-cloud-aiplatform==1.115.0 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core==0.23.0
- name: Setup litellm-enterprise
run: |
poetry run pip install --force-reinstall --no-deps -e enterprise/
- name: Generate Prisma client
env:
PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache
run: |
poetry run pip install nodejs-wheel-binaries==24.13.1
poetry run prisma generate --schema litellm/proxy/schema.prisma
# Run the same documentation tests that CircleCI ran (as direct Python scripts)
- name: Run documentation validation tests
run: |
poetry run python ./tests/documentation_tests/test_env_keys.py
poetry run python ./tests/documentation_tests/test_router_settings.py
poetry run python ./tests/documentation_tests/test_api_docs.py
poetry run python ./tests/documentation_tests/test_circular_imports.py

View file

@ -0,0 +1,24 @@
name: "Unit Tests: Enterprise, Google GenAI & Routing"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
enterprise-routing:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: >-
tests/test_litellm/enterprise
tests/test_litellm/google_genai
tests/test_litellm/router_utils
tests/test_litellm/router_strategy
workers: 2
reruns: 2

View file

@ -0,0 +1,20 @@
name: "Unit Tests: Integrations (Callbacks & Logging)"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
integrations:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/integrations"
workers: 2
reruns: 3

View file

@ -0,0 +1,29 @@
name: "Unit Tests: LLM Provider Transformations"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
vertex-ai:
name: Vertex AI
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/llms/vertex_ai"
workers: 1
reruns: 2
other-providers:
name: All Other Providers
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/llms --ignore=tests/test_litellm/llms/vertex_ai"
workers: 2
reruns: 2

31
.github/workflows/test-unit-misc.yml vendored Normal file
View file

@ -0,0 +1,31 @@
name: "Unit Tests: MCP, Secrets, Containers & Misc"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
misc:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: >-
tests/test_litellm/secret_managers
tests/test_litellm/a2a_protocol
tests/test_litellm/anthropic_interface
tests/test_litellm/completion_extras
tests/test_litellm/containers
tests/test_litellm/experimental_mcp_client
tests/test_litellm/images
tests/test_litellm/interactions
tests/test_litellm/passthrough
tests/test_litellm/vector_stores
tests/test_litellm/test_*.py
workers: 2
reruns: 2

View file

@ -0,0 +1,20 @@
name: "Unit Tests: Proxy Auth & Key Management"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
proxy-auth:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/proxy/auth tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine tests/test_litellm/proxy/client"
workers: 2
reruns: 2

View file

@ -0,0 +1,45 @@
name: "Unit Tests: Proxy DB Operations"
# Uses DATABASE_URL secret — only runs on trusted branches, not PRs.
on:
push:
branches: [main, "litellm_*"]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
proxy-db:
strategy:
fail-fast: false
matrix:
include:
# Key generation tests must NOT run in parallel (event loop conflicts with logging worker)
- test-group: key-generation
test-path: "tests/proxy_unit_tests/test_key_generate_prisma.py"
workers: 0
timeout: 30
- test-group: auth-checks
test-path: "tests/proxy_unit_tests/test_auth_checks.py tests/proxy_unit_tests/test_user_api_key_auth.py"
workers: 8
timeout: 20
- test-group: remaining
test-path: "tests/proxy_unit_tests --ignore=tests/proxy_unit_tests/test_key_generate_prisma.py --ignore=tests/proxy_unit_tests/test_auth_checks.py --ignore=tests/proxy_unit_tests/test_user_api_key_auth.py"
workers: 8
timeout: 20
uses: ./.github/workflows/_test-unit-services-base.yml
with:
test-path: ${{ matrix.test-path }}
workers: ${{ matrix.workers }}
reruns: 2
timeout-minutes: ${{ matrix.timeout }}
enable-redis: false
enable-postgres: true
secrets:
DATABASE_URL: ${{ secrets.DATABASE_URL }}
POSTGRES_USER: ${{ secrets.POSTGRES_USER }}
POSTGRES_PASSWORD: ${{ secrets.POSTGRES_PASSWORD }}

View file

@ -0,0 +1,35 @@
name: "Unit Tests: Proxy API Endpoints"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
proxy-endpoints:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: >-
tests/test_litellm/proxy/management_endpoints
tests/test_litellm/proxy/guardrails
tests/test_litellm/proxy/management_helpers
tests/test_litellm/proxy/anthropic_endpoints
tests/test_litellm/proxy/google_endpoints
tests/test_litellm/proxy/openai_files_endpoint
tests/test_litellm/proxy/response_api_endpoints
tests/test_litellm/proxy/image_endpoints
tests/test_litellm/proxy/vector_store_endpoints
tests/test_litellm/proxy/agent_endpoints
tests/test_litellm/proxy/discovery_endpoints
tests/test_litellm/proxy/health_endpoints
tests/test_litellm/proxy/public_endpoints
tests/test_litellm/proxy/prompts
tests/test_litellm/proxy/ui_crud_endpoints
workers: 2
reruns: 2

View file

@ -0,0 +1,28 @@
name: "Unit Tests: Proxy Infrastructure"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
proxy-infra:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: >-
tests/test_litellm/proxy/db
tests/test_litellm/proxy/middleware
tests/test_litellm/proxy/spend_tracking
tests/test_litellm/proxy/pass_through_endpoints
tests/test_litellm/proxy/_experimental
tests/test_litellm/proxy/experimental
tests/test_litellm/proxy/common_utils
tests/test_litellm/proxy/test_*.py
workers: 2
reruns: 2

View file

@ -0,0 +1,96 @@
name: "Unit Tests: Proxy Legacy Tests"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
test:
runs-on: ubuntu-latest
timeout-minutes: 20
strategy:
fail-fast: false
matrix:
test-group:
- name: "auth-and-jwt"
path: "tests/proxy_unit_tests/test_[a-j]*.py"
- name: "key-generation"
path: "tests/proxy_unit_tests/test_[k-o]*.py"
- name: "proxy-config"
path: "tests/proxy_unit_tests/test_prisma*.py tests/proxy_unit_tests/test_project*.py tests/proxy_unit_tests/test_prompt*.py tests/proxy_unit_tests/test_proxy_[c-r]*.py"
- name: "proxy-server"
path: "tests/proxy_unit_tests/test_proxy_server.py"
- name: "proxy-server-extras"
path: "tests/proxy_unit_tests/test_proxy_server_*.py tests/proxy_unit_tests/test_proxy_setting_guardrails.py"
- name: "proxy-utils"
path: "tests/proxy_unit_tests/test_proxy_utils.py"
- name: "proxy-token-counter"
path: "tests/proxy_unit_tests/test_proxy_token_counter.py"
- name: "proxy-response-and-misc"
path: "tests/proxy_unit_tests/test_[r-t]*.py"
- name: "proxy-user-auth-and-spend"
path: "tests/proxy_unit_tests/test_[u-z]*.py"
name: ${{ matrix.test-group.name }}
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.12"
- name: Install Poetry
run: pip install 'poetry==2.3.2'
- name: Cache Poetry dependencies
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/pypoetry
~/.cache/pip
.venv
key: ${{ runner.os }}-poetry-${{ hashFiles('poetry.lock') }}
restore-keys: |
${{ runner.os }}-poetry-
- name: Install dependencies
run: |
poetry config virtualenvs.in-project true
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install google-genai==1.22.0 \
google-cloud-aiplatform==1.115.0 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core==0.23.0
- name: Setup litellm-enterprise
run: |
poetry run pip install --force-reinstall --no-deps -e enterprise/
- name: Generate Prisma client
env:
PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache
run: |
poetry run pip install nodejs-wheel-binaries==24.13.1
poetry run prisma generate --schema litellm/proxy/schema.prisma
- name: Run tests - ${{ matrix.test-group.name }}
env:
TEST_PATH: ${{ matrix.test-group.path }}
run: |
poetry run pytest ${TEST_PATH} \
--tb=short -vv \
--maxfail=10 \
-n 2 \
--reruns 1 \
--reruns-delay 1 \
--dist=loadscope \
--durations=20

View file

@ -0,0 +1,20 @@
name: "Unit Tests: Responses, Caching & Types"
on:
pull_request:
branches: [main]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
responses-caching-types:
uses: ./.github/workflows/_test-unit-base.yml
with:
test-path: "tests/test_litellm/responses tests/test_litellm/caching tests/test_litellm/types"
workers: 2
reruns: 2

View file

@ -0,0 +1,28 @@
name: "Unit Tests: Security"
# Uses DATABASE_URL secret — only runs on trusted branches, not PRs.
on:
push:
branches: [main, "litellm_*"]
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
security:
uses: ./.github/workflows/_test-unit-services-base.yml
with:
test-path: "tests/proxy_security_tests/"
workers: 1
reruns: 2
timeout-minutes: 20
enable-redis: false
enable-postgres: true
secrets:
DATABASE_URL: ${{ secrets.DATABASE_URL }}
POSTGRES_USER: ${{ secrets.POSTGRES_USER }}
POSTGRES_PASSWORD: ${{ secrets.POSTGRES_PASSWORD }}

31
.github/workflows/zizmor.yml vendored Normal file
View file

@ -0,0 +1,31 @@
name: GitHub Actions Security Analysis
on:
push:
branches: [main]
pull_request:
branches: [main]
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
permissions: {}
jobs:
zizmor:
name: zizmor
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
security-events: write
contents: read
actions: read
steps:
- name: Checkout repository
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Run zizmor
uses: zizmorcore/zizmor-action@71321a20a9ded102f6e9ce5718a2fcec2c4f70d8 # v0.5.2

3
.gitignore vendored
View file

@ -72,8 +72,7 @@ tests/local_testing/log.txt
.codegpt
litellm/proxy/_new_new_secret_config.yaml
litellm/proxy/custom_guardrail.py
.mypy_cache/*
.mypy_cache/*
**/.mypy_cache/
litellm/proxy/application.log
tests/llm_translation/vertex_test_account.json
tests/llm_translation/test_vertex_key.json

5
.npmrc Normal file
View file

@ -0,0 +1,5 @@
# Supply-chain hardening
# Packages needing lifecycle scripts: npm rebuild <pkg>
ignore-scripts=true
# Protects local npm install only — npm ci (used in CI) ignores this
min-release-age=3d

View file

@ -266,6 +266,7 @@ Support for more providers. Missing a provider or LLM Platform, raise a [feature
<table>
<tr>
<td><img height="60" alt="Stripe" src="https://github.com/user-attachments/assets/f7296d4f-9fbd-460d-9d05-e4df31697c4b" /></td>
<td><img height="60" alt="image" src="https://github.com/user-attachments/assets/436fca71-988b-40bb-b5fe-8450c80fdbd0" /></td>
<td><img height="60" alt="Google ADK" src="https://github.com/user-attachments/assets/caf270a2-5aee-45c4-8222-41a2070c4f19" /></td>
<td><img height="60" alt="Greptile" src="https://github.com/user-attachments/assets/0be4bd8a-7cfa-48d3-9090-f415fe948280" /></td>
<td><img height="60" alt="OpenHands" src="https://github.com/user-attachments/assets/a6150c4c-149e-4cae-888b-8b92be6e003f" /></td>

View file

@ -160,7 +160,7 @@ run_grype_scans() {
"CVE-2026-0775" # npm cli incorrect permission assignment - no fix available yet, npm is only used at build/prisma-generate time
"GHSA-3ppc-4f35-3m26" # minimatch ReDoS via repeated wildcards - from nodejs_wheel bundled npm, not used in application runtime code
"GHSA-83g3-92jg-28cx" # tar arbitrary file read/write via hardlink - from nodejs_wheel bundled npm, not used in application runtime code
"CVE-2026-25639" # axios - full fix requires 1.x major version bump; pinned to >=0.30.2 to clear other axios CVEs, upgrade to 1.x in follow-up
"CVE-2026-25639" # axios DoS via __proto__ in mergeConfig - transitive dev dep via @neondatabase/api-client, not imported in application code
"CVE-2026-2297" # Python 3.13 SourcelessFileLoader audit hook bypass - no fix available in base image
"GHSA-qffp-2rhf-9h96" # tar hardlink path traversal - from nodejs_wheel bundled npm, not used in application runtime code
"CVE-2026-2673" # OpenSSL 3.6.1 TLS 1.3 key exchange group negotiation issue - no fix available yet

View file

@ -17,6 +17,9 @@ component_management:
- component_id: "Proxy_Authentication"
paths:
- "*/proxy/auth/**"
- component_id: "Enterprise"
paths:
- "enterprise/**"
comment:
layout: "header, diff, flags, components" # show component info in the PR comment

4
cosign.pub Normal file
View file

@ -0,0 +1,4 @@
-----BEGIN PUBLIC KEY-----
MFkwEwYHKoZIzj0CAQYIKoZIzj0DAQcDQgAEKi4ivqGpE231OGH50PKbqy1Y1Kkb
POJC8+i2Wko82gBOUCe3M0Vw86H/4rhUhfoYEti4gdJ9wZbYmK0I2EE96g==
-----END PUBLIC KEY-----

View file

@ -51,7 +51,7 @@ ENV UI_BASE_PATH="/prod/ui"
# Build the UI with the specified UI_BASE_PATH
WORKDIR /app/ui/litellm-dashboard
RUN npm install
RUN npm ci
RUN UI_BASE_PATH=$UI_BASE_PATH npm run build
# Create the destination directory

View file

@ -47,7 +47,7 @@ RUN mkdir -p /var/lib/litellm/ui && \
if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
fi && \
npm install --legacy-peer-deps && \
npm ci && \
npm run build && \
cp -r /app/ui/litellm-dashboard/out/* /var/lib/litellm/ui/ && \
mkdir -p /var/lib/litellm/assets && \

View file

@ -40,11 +40,22 @@ else
exit 1
fi
fi
curl -o- https://raw.githubusercontent.com/nvm-sh/nvm/v0.38.0/install.sh | bash
NVM_VERSION="v0.40.4"
NVM_CHECKSUM="4b7412c49960c7d31e8df72da90c1fb5b8cccb419ac99537b737028d497aba4f"
NVM_SCRIPT=$(mktemp)
trap 'rm -f "$NVM_SCRIPT"' EXIT
curl -fsSL "https://raw.githubusercontent.com/nvm-sh/nvm/${NVM_VERSION}/install.sh" -o "$NVM_SCRIPT"
if command -v sha256sum &>/dev/null; then
echo "${NVM_CHECKSUM} ${NVM_SCRIPT}" | sha256sum -c -
elif command -v shasum &>/dev/null; then
echo "${NVM_CHECKSUM} ${NVM_SCRIPT}" | shasum -a 256 -c -
else
echo "No sha256 tool found; cannot verify nvm checksum"; exit 1
fi || { echo "nvm checksum verification failed"; exit 1; }
bash "$NVM_SCRIPT"
source ~/.nvm/nvm.sh
nvm install v18.17.0
nvm use v18.17.0
npm install -g npm
# copy _enterprise.json from this directory to /ui/litellm-dashboard, and rename it to ui_colors.json
cp enterprise/enterprise_ui/enterprise_colors.json ui/litellm-dashboard/ui_colors.json

View file

@ -0,0 +1,55 @@
---
slug: ci-cd-v2-improvements
title: "Announcing CI/CD v2 for LiteLLM"
date: 2026-03-30T21:30:00
authors:
- krrish
description: "CI/CD v2 introduces isolated environments, stronger security gates, and safer release separation for LiteLLM."
tags: [engineering, ci-cd, security]
hide_table_of_contents: false
---
import Image from '@theme/IdealImage';
The CI/CD v2 is now live for LiteLLM.
<Image
img={require('../../img/ci_cd_architecture.png')}
style={{width: '700px', height: 'auto', display: 'block'}}
/>
<br/>
Building on the roadmap from our [security incident](https://docs.litellm.ai/blog/security-townhall-updates#roadmap), CI/CD v2 introduces isolated environments, stronger security gates, and safer release separation for LiteLLM.
## What changed
- Security scans and unit tests run in isolated environments.
- Validation and release are separated into different repositories, making it harder for an attacker to reach release credentials.
- Trusted Publishing for PyPI releases - this means no long-lived credentials are used to publish releases.
- Immutable Docker release tags - this means no tampering of Docker release tags after they are published [Learn more](https://docs.docker.com/docker-hub/repos/manage/hub-images/immutable-tags/). Note: work for GHCR docker releases is planned as well.
## What's next
Moving forward, we plan on:
- Adopting OpenSSF (this is a set of security criteria that projects should meet to demonstrate a strong security posture - [Learn more](https://baseline.openssf.org/versions/2026-02-19.html))
- We've added Scorecard and Allstar to our Github
- Adding SLSA Build Provenance to our CI/CD pipeline - this means we allow users to independently verify that a release came from us and prevent silent modifications of releases after they are published.
We hope that this will mean you can be confident that the releases you are using are safe and from us.
## The principle
The new CI/CD pipeline reflects the principles, outlined below, and is designed to be more secure and reliable:
- **Limit** what each package can access
- **Reduce** the number of sensitive environment variables
- **Avoid** compromised packages
- **Prevent** release tampering
## How to help:
Help us plan April's stability sprint - https://github.com/BerriAI/litellm/issues/24825

View file

@ -0,0 +1,190 @@
---
slug: security-townhall-updates
title: "Security Townhall Updates"
date: 2026-03-27T12:00:00
authors:
- krrish
- ishaan-alt
description: "What happened, what we've done, and what comes next for LiteLLM's release and security processes."
tags: [security, incident-report]
hide_table_of_contents: false
---
import Image from '@theme/IdealImage';
Thank you to everyone who joined our town hall.
We wanted to use that time to walk through what we know, what we've done so far, and how we're improving LiteLLM's release and security processes going forward. This post is a written version of that update. [Slides available here](https://drive.google.com/file/d/17hsSG7nk-OYL7VRCTbTa7McrWREtS9OO/view?usp=sharing)
{/* truncate */}
## What happened
On March 24, 2026 at 10:39 UTC, LiteLLM v1.82.7 was pushed to PyPI. Version v1.82.8 was published soon after. Those packages were live for about 40 minutes before being quarantined by PyPI. By 16:00 UTC, the LiteLLM team had worked with PyPI to delete the affected packages.
At this point, our understanding is that this was a supply-chain incident affecting those two published versions.
## How did this happen?
Our understanding is that the issue came from the [compromised Trivy security scanner](https://www.aquasec.com/blog/trivy-supply-chain-attack-what-you-need-to-know/) dependency in our CI/CD pipeline.
<Image
img={require('../../img/shared_ci_cd_environment.png')}
style={{width: '500px', height: '400px', display: 'block'}}
/>
There were three major contributing factors:
### 1. Shared CI/CD environment
At the time, everything was running on CircleCI, and all steps shared a common environment. That increased blast radius: if one component was compromised, it could potentially access credentials or context intended for other parts of the pipeline.
### 2. Static credentials in environment variables
Release credentials, including credentials for PyPI, GHCR, and Docker publishing, were available as static secrets in the environment. That meant a compromised step could access long-lived release credentials.
### 3. Unpinned Trivy dependency
In our security scanning component, we had an unpinned Trivy dependency. Our present understanding is that a compromised Trivy package ran during the scan, had access to environment variables, and enabled attackers to obtain those credentials.
**In summary:** a compromised package in CI had access to secrets it should not have had, and those secrets were then used in the release path.
## What we've already done
In the last 3 days, we've taken the following steps:
### 1. Minimize Scope of Impact
#### Prevented further key abuse
We deleted or rotated all impacted or adjacent secret keys, including PyPI, GitHub, Docker, and related credentials. Out of an abundance of caution, we've also rotated LiteLLM maintainer accounts.
#### Prevent branch attacks
We removed roughly 6,000 open branches and added an auto-deletion policy for branches merged into `main`. This reduces the surface area for branch-based abuse.
#### Pinned CI/CD dependencies
We've pinned all Github Actions, and are working on pinning all CircleCI dependencies as well.
#### Paused releases
We've paused new releases until we've confirmed codebase security and put stronger release controls in place.
### 2. Secured LiteLLM
#### Forensic analysis
We are working with Google's Mandiant cybersecurity team to confirm the source of the attack and verify the security of the codebase. We also confirmed that no malicious code was pushed to `main`.
#### Confirm Application Security
In parallel, we are working with whitehat hackers at [Veria Labs](https://verialabs.com/) to verify application security and review improvements to our CI/CD process.
We have also confirmed that the last 20 LiteLLM releases contain no indicators of compromise, and that no unauthenticated attacks can be made against LiteLLM Proxy based on our current investigation. [Check Security Blog for release verification.](https://docs.litellm.ai/blog/security-update-march-2026#verified-safe-versions)
#### Created a security working group
We created a new security working group inside LiteLLM focused on:
- Building threat models
- Auditing the build process and dependencies
If you're interested in joining the security working group, please file an issue [here](https://github.com/BerriAI/litellm-security-wg).
### 3. Improved CI/CD
We've already begun making structural changes to how releases are built and published. These align with our goals (covered in the next section) around isolated environments, ephemeral credentials, and release auditing.
## Roadmap
We plan on following 4 guiding principles for our new CI/CD pipeline:
1. **Limit** what each package can access
2. **Reduce** the number of sensitive environment variables
3. **Avoid** compromised packages
4. **Prevent** release tampering
### Isolated environments
<Image
img={require('../../img/isolated_ci_cd_environments.png')}
style={{width: '400px', height: 'auto'}}
/>
We are breaking our CI/CD into 4 semantic concepts:
1. Unit tests
2. Integration tests
3. Security scans
4. Release publishing
And will be running each of these in isolated environments.
This will limit the damage that any single compromised component can cause.
### Ephemeral credentials
We plan to move to ephemeral credentials for PyPI (Trusted Publisher) and GHCR (Token-based authentication) releases. This will reduce the risk of credentials being leaked or compromised.
We have already begun doing this:
- PyPI Trusted Publisher on GitHub Actions [PR](https://github.com/BerriAI/litellm/pull/24654)
- GHCR Token-based authentication on GitHub Actions [PR](https://github.com/BerriAI/litellm/pull/24683)
### Release auditing
Our goal is to allow users to independently verify that a release came from us and prevent silent modifications of releases after they are published.
This will ensure, your releases are safe, even when:
- Stolen PyPI/GHCR credentials are used to publish malicious releases
- Tampered registry artifacts are published
- Tag mutations are made after the release is published
We believe that [Cosign](https://github.com/sigstore/cosign) is a good fit for this, and have already begun working on it [PR](https://github.com/BerriAI/litellm/pull/24683).
### Avoid Compromised Packages
- Move to pinned, verified SHAs for packages and actions used in CI/CD, avoiding `latest` wherever possible.
- Add a cooldown period before upgrading to a new version of a package - allows more time to investigate and verify the new version.
We've added zizmor to help us catch issues such as unpinned dependencies and credential leakage. [commit](https://github.com/BerriAI/litellm/commit/a671275f5c5b0e1fb1adacdf3b6ef779aaa5d56c).
## Frequently Asked Questions
**Q: Did you observe any lateral movement into your corporate environment during this incident?**
A: No. Our investigation to date, conducted in coordination with external security experts, has found no evidence of lateral movement into our internal corporate systems. The incident was isolated to the CI/CD pipeline and the release path for specific versions (v1.82.7 and v1.82.8). As a proactive measure, we have rotated all potentially impacted or adjacent secrets—including PyPI, GitHub, and Docker credentials—and updated maintainer account security to ensure continued isolation.
**Q: Do you expect delays in future product releases due to these new security measures?**
A: We are committed to balancing security with speed. While we have temporarily paused releases to implement stronger controls, we are moving quickly to automate our new security protocols. We are currently implementing isolated CI/CD environments, ephemeral credentials (via Trusted Publishers), and release auditing with Cosign. These improvements are designed to be integrated into our automated pipeline, allowing us to maintain a fast release cadence while ensuring every package is verified and secure.
**Q: Were older packages impacted?**
Our current findings show no indicators of compromise in the last 20 versions of LiteLLM. This was manually verified by our team and independently reviewed by Veria Labs.
We have also published the verified versions for users to use. [Check Security Blog for release verification.](https://docs.litellm.ai/blog/security-update-march-2026#verified-safe-versions)
## Questions & Support
If you believe your systems may be affected, contact us immediately:
- **Security:** security@berri.ai
- **Support:** support@berri.ai
- **Slack:** Reach out to the LiteLLM team directly [here](https://join.slack.com/t/litellmossslack/shared_invite/zt-3o7nkuyfr-p_kbNJj8taRfXGgQI1~YyA)
## Hiring
We are currently hiring for:
- DevOps Engineer - to keep ci/cd secure and running smoothly
- Security Engineer - to keep the application secure
If you're interest in joining, please apply [here](https://jobs.ashbyhq.com/litellm)

Binary file not shown.

After

Width:  |  Height:  |  Size: 45 KiB

View file

@ -12,18 +12,27 @@ hide_table_of_contents: false
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import VersionVerificationTable from '@site/src/components/VersionVerificationTable';
> **Status:** Active investigation
> **Last updated:** March 25, 2026
> **Last updated:** March 27, 2026
> **Update (March 30):** A new **clean** version of LiteLLM is now available (v1.83.0). This was released by our new [CI/CD v2](https://docs.litellm.ai/blog/ci-cd-v2-improvements) pipeline which added isolated environments, stronger security gates, and safer release separation for LiteLLM.
> **Update (March 27):** Review Townhall updates, including explanation of the incident, what we've done, and what comes next. [Learn more](https://docs.litellm.ai/blog/security-townhall-updates)
> **Update (March 27):** Added [Verified safe versions](#verified-safe-versions) section with SHA-256 checksums for all audited PyPI and Docker releases.
> **Update (March 26):** Added `checkmarx[.]zone` to [Indicators of compromise](#indicators-of-compromise-iocs)
> **Update (March 25):** Added community-contributed scripts for scanning GitHub Actions and GitLab CI pipelines for the compromised versions. See [How to check if you are affected](#how-to-check-if-you-are-affected). s/o [@Zach Fury](https://www.linkedin.com/in/fryware/) for these scripts.
## TLDR;
- The compromised PyPI packages were **litellm==1.82.7** and **litellm==1.82.8**. Those packages have now been removed from PyPI.
- We believe that the compromise originated from the Trivy dependency used in our CI/CD security scanning workflow.
- The compromised PyPI packages were **litellm==1.82.7** and **litellm==1.82.8**. Those packages were live on March 24, 2026 from 10:39 UTC for about 40 minutes before being quarantined by PyPI.
- We believe that the compromise originated from the [Trivy dependency](https://www.aquasec.com/blog/trivy-supply-chain-attack-what-you-need-to-know/) used in our CI/CD security scanning workflow.
- Customers running the official LiteLLM Proxy Docker image were not impacted. That deployment path pins dependencies in requirements.txt and does not rely on the compromised PyPI packages.
- We are pausing new LiteLLM releases until we complete a broader supply-chain review and confirm the release path is safe.
- ~~We have paused all new LiteLLM releases until we complete a broader supply-chain review and confirm the release path is safe.~~ **Updated:** We have now released a new **safe** version of LiteLLM (v1.83.0) by our new [CI/CD v2](https://docs.litellm.ai/blog/ci-cd-v2-improvements) pipeline which added isolated environments, stronger security gates, and safer release separation for LiteLLM. We have also verified the codebase is safe and no malicious code was pushed to `main`.
## Overview
@ -643,6 +652,8 @@ Review affected systems for the following indicators:
- `litellm_init.pth` present in your `site-packages`
- Outbound traffic or requests to `models.litellm[.]cloud`
This domain is **not** affiliated with LiteLLM
- Outbound traffic or requests to `checkmarx[.]zone`
This domain is **not** affiliated with LiteLLM
## Immediate actions for affected users
@ -697,6 +708,72 @@ The LiteLLM AI Gateway team has already taken the following steps:
- Engaged Google's Mandiant security team to assist with forensic analysis of the build and publishing chain
## Verified safe versions
We have audited every LiteLLM release published between v1.78.0 and v1.82.6 across both PyPI and Docker. Each artifact was verified by:
1. Downloading the published artifact and computing its SHA-256 digest
2. Scanning for the known [indicators of compromise](#indicators-of-compromise-iocs) (IOCs)
3. Comparing the artifact contents against the corresponding Git commit in the BerriAI/litellm repository
**All versions listed below are confirmed clean.**
<Tabs>
<TabItem value="pypi" label="PyPI Releases">
<VersionVerificationTable entries={[
{ version: "1.82.6", sha256: "164a3ef3e19f309e3cabc199bef3d2045212712fefdfa25fc7f75884a5b5b205", gitCommit: "38d477507dad" },
{ version: "1.82.5", sha256: "e1012ab816352215c4e00776dd48b0c68058b537888a8ff82cca62af19e6fb11", gitCommit: "1998c4f3703f" },
{ version: "1.82.4", sha256: "d37c34a847e7952a146ed0e2888a24d3edec7787955c6826337395e755ad5c4b", gitCommit: "cfeafbe38811" },
{ version: "1.82.3", sha256: "609901f6c5a5cf8c24386e4e3f50738bb8a9db719709fd76b208c8ee6d00f7a7", gitCommit: "61409275c8d8" },
{ version: "1.82.2", sha256: "641ed024774fa3d5b4dd9347f0efb1e31fa422fba2a6500aabedee085d1194cb", gitCommit: "f351bbdb3683" },
{ version: "1.82.1", sha256: "a9ec3fe42eccb1611883caaf8b1bf33c9f4e12163f94c7d1004095b14c379eb2", gitCommit: "94b002066e3a" },
{ version: "1.82.0", sha256: "5496b5d4532cccdc7a095c21cbac4042f7662021c57bc1d17be4e39838929e80", gitCommit: "6c6585af568e" },
{ version: "1.81.16", sha256: "d6bcc13acbd26719e07bfa6b9923740e88409cbf1f9d626d85fc9ae0e0eec88c", gitCommit: "678200ee4887" },
{ version: "1.81.15", sha256: "2fa253658702509ce09fe0e172e5a47baaadf697fb0f784c7fd4ff665ae76ae1", gitCommit: "2e819656cee9" },
{ version: "1.81.14", sha256: "6394e61bbdef7121e5e3800349f6b01e9369e7cf611e034f1832750c481abfed", gitCommit: "96bcee0b0af7" },
{ version: "1.81.13", sha256: "ae4aea2a55e85993f5f6dd36d036519422d24812a1a3e8540d9e987f2d7a4304", gitCommit: "cc957a19a560" },
{ version: "1.81.12", sha256: "219cf9729e5ea30c6d3f75aa43fef3c56a717369939a6d717cbad0fd78e3c146", gitCommit: "ba0d541b1982" },
{ version: "1.81.11", sha256: "06a66c24742e082ddd2813c87f40f5c12fe7baa73ce1f9457eaf453dc44a0f65", gitCommit: "231aedeeff7e" },
{ version: "1.81.10", sha256: "9efa1cbe61ac051f6500c267b173d988ff2d511c2eecf1c8f2ee546c0870747c", gitCommit: "7488abece8e7" },
{ version: "1.81.9", sha256: "24ee273bc8a62299fbb754035f83fb7d8d44329c383701a2bd034f4fd1c19084", gitCommit: "a09d3e9162eb" },
{ version: "1.81.8", sha256: "78cca92f36bc6c267c191d1fe1e2630c812bff6daec32c58cade75748c2692f6", gitCommit: "4fea649f519b" },
{ version: "1.81.7", sha256: "58466c88c3289c6a3830d88768cf8f307581d9e6c87861de874d1128bb2de90d", gitCommit: "3f6a281d0f7a" },
{ version: "1.81.6", sha256: "573206ba194d49a1691370ba33f781671609ac77c35347f8a0411d852cf6341a", gitCommit: "8da3a93e6e63" },
{ version: "1.81.5", sha256: "206505c5a0c6503e465154b9c979772be3ede3f5bf746d15b37dca5ae54d239f", gitCommit: "2cc3778761d4" },
{ version: "1.81.3", sha256: "3f60fd8b727587952ad3dd18b68f5fed538d6f43d15bb0356f4c3a11bccb2b92", gitCommit: "f30742fe6e8e" },
]} />
</TabItem>
<TabItem value="docker" label="Docker Images">
<VersionVerificationTable entries={[
{ version: "1.82.3", sha256: "0a571da849db5f9c3cf3fead2ffbf1df982eebff7e7b38b46dbec3f640dafdbb", gitCommit: "61409275c8d8" },
{ version: "1.82.3-stable", sha256: "0c2b2a0ad3e50af1702fc493ecd07f22a5180b6d1cfb169440b429b40e340e29", gitCommit: "61409275c8d8" },
{ version: "1.82.0-stable", sha256: "71bf7283767ca436edcfa9f1f26c1743487b5fa29736c61c3eb6977776007c42", gitCommit: "97947c254252" },
{ version: "1.81.15", sha256: "303c31af87e7915e7b34d6c4d55a6ac753ef947a5deaa899e9ccfd3d1d58f7c2", gitCommit: "20bf3aa8070a" },
{ version: "1.81.14-stable", sha256: "a34f9758048231817d799b703fb998e40e2a5cbabb89ab95039fc30798f01b3c", gitCommit: "0435375b1271" },
{ version: "1.81.13", sha256: "a876f3f22f9b6fd481c9091c44a8a893d81c172d66dc2749298dcd3dc4a3d6f0", gitCommit: "cc957a19a560" },
{ version: "1.81.12-stable", sha256: "e24022878ccc87f57d808ac9304f18b87b8359e6556746d81cc20a5dc85f423a", gitCommit: "ba0d541b1982" },
{ version: "1.81.9-stable", sha256: "262e53d7702ed82579717faff0b08f7c0b7e9973a6406cfcc0e4af7826327627", gitCommit: "a09d3e9162eb" },
{ version: "1.81.3-stable", sha256: "dff82ccc32fb648927c090607887401c7e8ec814fe7c951beb95fe51073ca02b", gitCommit: "61ed8f9e0355" },
{ version: "1.81.0-stable", sha256: "f4913297d1bb3dc373eb8911a5ac816b597be9b5e08a91636b6c2786dd572aa8", gitCommit: "790a5ce0b323" },
{ version: "1.80.15-stable", sha256: "0b4ec3861e978b4aa254f4070f292cd345496a5fb59c72e1ee21cd6db94b670b", gitCommit: "17c8d8d109b5" },
{ version: "1.80.11-stable", sha256: "4068108d9101cd2affba3924310fd7f34f23d14e36dd4853733898b9e04d81ca", gitCommit: "57e07bddd341" },
{ version: "1.80.8-stable", sha256: "0304c2eb1f3cf54262d1b4e0629487232bab459e95b99a21e5810231d2b27021", gitCommit: "3381d63152f8" },
{ version: "1.80.5-stable", sha256: "a89e173135fff96af4b5b91ea31845164eadcf6497c82adeb64c36a23c8a3d11", gitCommit: "6c49b95a4ab7" },
{ version: "1.80.0-stable", sha256: "a3416f4cd0c896c94a1f526d872ff6c19bee22ff4afcdcc6f9ff690707900176", gitCommit: "98365205acd0" },
{ version: "1.79.3-stable", sha256: "27aae83d6ab6cb0b63bf8179e375ce0e11f5cfef51f2675b0c1e60c6f546dbc1", gitCommit: "c0548542d4a9" },
{ version: "1.79.1-stable", sha256: "7780d29a9543c4ce762430db7dfb0640105f7357fc38e35bf3fb7bbb1e6ba63f", gitCommit: "c217bddb59ba" },
{ version: "1.79.0-stable", sha256: "32bf6ac059a56641e11e4712f63b8467c295f988b6c160dc7229660417ee44bd", gitCommit: "8d495f56a9cc" },
{ version: "1.78.5-stable", sha256: "d5e607648eafa15edc63b0b1a5ed01f8b31a1fa0c80f7d25b252ae18a593ee29", gitCommit: "c471bf1f16c2" },
{ version: "1.78.0-stable", sha256: "7a56b32dc7153763d31c0a056123dc878a598959935d8c7daacb1fca5272c205", gitCommit: "5fde83d9f154" },
]} />
</TabItem>
</Tabs>
## Questions and support
If you believe your systems may be affected, contact us immediately:

View file

@ -0,0 +1,18 @@
---
slug: vanta-compliance-recertification
title: "LiteLLM + Vanta: SOC 2 Type 2 and ISO 27001 Recertification"
date: 2026-03-30T10:00:00
authors:
- krrish
description: "LiteLLM is partnering with Vanta on SOC 2 Type 2 and ISO 27001 recertification and engaging independent auditors for verification."
tags: [security, compliance]
hide_table_of_contents: true
---
![LiteLLM x Vanta SOC-2 Recertification](/img/blog/vanta_soc2_recertification.png)
We are partnering with [Vanta](https://www.vanta.com/) to recertify LiteLLM's compliance for SOC 2 Type 2 and ISO 27001.
As part of this process, we are also identifying independent auditors to validate and verify our compliance posture.
This is part of our commitment to being the most secure and transparent AI Gateway possible.

View file

@ -11,7 +11,7 @@ Use this path if you are integrating LiteLLM directly into application code.
## 1. Install LiteLLM
```bash
pip install litellm
pip install litellm==1.82.6
```
## 2. Set Provider Credentials

View file

@ -11,6 +11,7 @@ import TabItem from '@theme/TabItem';
| Provider Doc | [Google AI Studio ↗](https://aistudio.google.com/) |
| API Endpoint for Provider | https://generativelanguage.googleapis.com |
| Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions`, [`/videos`](./gemini/videos.md), [`/images/edits`](../image_edits.md) |
| Lyria (music) | [Cost map & notes](./gemini/music.md) |
| Pass-through Endpoint | [Supported](../pass_through/google_ai_studio.md) |
<br />

View file

@ -0,0 +1,28 @@
# Gemini — Lyria (music generation)
Google Lyria 3 preview models are listed in LiteLLM’s [model cost map](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json) under the `gemini/` provider for metadata and spend tracking.
| Property | Details |
|----------|---------|
| Provider route | `gemini/` |
| Models | `gemini/lyria-3-clip-preview`, `gemini/lyria-3-pro-preview` |
| Provider docs | [Gemini API pricing / models ↗](https://ai.google.dev/gemini-api/docs/pricing) |
## Models
| Model | Notes |
|-------|--------|
| `gemini/lyria-3-clip-preview` | ~30s clip; paid tier listed as per generated song in Google’s pricing |
| `gemini/lyria-3-pro-preview` | Full song; paid tier listed as per generated song in Google’s pricing |
Input context limit in the cost map: **131,072** tokens. For modalities, limits, and features, see [Google’s Gemini API docs ↗](https://ai.google.dev/gemini-api/docs/models).
## LiteLLM behavior
- **Cost map**: Per-song paid pricing is stored as `output_cost_per_image` on those entries (flat per generation unit). Token-based completion cost may not reflect music billing until a dedicated path exists.
- **API calls**: Use the Gemini API as documented by Google. LiteLLM does not ship a separate `music_generation` helper like Veo’s `video_generation`.
## Auth
Same as other Gemini API models: `GEMINI_API_KEY` or `GOOGLE_API_KEY`.

View file

@ -581,6 +581,90 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
See [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning) for more details on organization verification requirements.
### Multi-turn Conversations with `reasoning_items`
For multi-turn conversations you need `reasoning_items`: structured blocks that include the `encrypted_content` token OpenAI uses to restore reasoning state on the next request. Pass `include=["reasoning.encrypted_content"]` on every call where you want that token returned.
<Tabs>
<TabItem value="non-streaming" label="Non-Streaming">
```python showLineNumbers title="Non-streaming: round-trip reasoning_items"
import litellm
messages = [{"role": "user", "content": "Solve this step by step: 2 + 2"}]
# Turn 1 — get reasoning_items (encrypted_content);
response = litellm.completion(
model="openai/responses/gpt-5-mini",
messages=messages,
reasoning_effort="low",
include=["reasoning.encrypted_content"],
)
assistant_msg = response.choices[0].message
# Turn 2 — pass reasoning_items back; LiteLLM converts to the correct Responses API format
messages.append({
"role": "assistant",
"content": assistant_msg.content,
"reasoning_items": assistant_msg.reasoning_items,
})
messages.append({"role": "user", "content": "Now summarize your reasoning."})
response2 = litellm.completion(
model="openai/responses/gpt-5-mini",
messages=messages,
reasoning_effort="low",
include=["reasoning.encrypted_content"],
)
```
</TabItem>
<TabItem value="streaming" label="Streaming">
`reasoning_items` (with `encrypted_content`) arrive on the final chunk when the full response completes:
```python showLineNumbers title="Streaming: collect and round-trip reasoning_items"
import litellm
messages = [{"role": "user", "content": "Solve this step by step: 2 + 2"}]
collected_content = []
collected_reasoning_items = []
stream = litellm.completion(
model="openai/responses/gpt-5-mini",
messages=messages,
stream=True,
reasoning_effort="low",
include=["reasoning.encrypted_content"],
)
for chunk in stream:
delta = chunk.choices[0].delta
if delta.content:
collected_content.append(delta.content)
if getattr(delta, "reasoning_items", None):
collected_reasoning_items.extend(delta.reasoning_items)
messages.append({
"role": "assistant",
"content": "".join(collected_content),
"reasoning_items": collected_reasoning_items or None,
})
messages.append({"role": "user", "content": "Continue the conversation."})
response2 = litellm.completion(
model="openai/responses/gpt-5-mini",
messages=messages,
reasoning_effort="low",
include=["reasoning.encrypted_content"],
)
```
</TabItem>
</Tabs>
### Verbosity Control for GPT-5 Models
The `verbosity` parameter controls the length and detail of responses from GPT-5 family models. It accepts three values: `"low"`, `"medium"`, or `"high"`.

View file

@ -279,6 +279,33 @@ router_settings:
| forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers and `anthropic-beta` headers) to the backend LLM call |
| maximum_spend_logs_retention_period | str | Used to set the max retention time for spend logs in the db, after which they will be auto-purged |
| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. |
| alert_type_config | dict | Configuration mapping alert types to their handler settings |
| always_include_stream_usage | boolean | If true, includes usage metrics in every streaming response chunk |
| auto_redirect_ui_login_to_sso | boolean | If true, automatically redirects UI login page to SSO provider |
| control_plane_url | string | URL of the control plane for cross-instance state sharing |
| custom_auth_run_common_checks | boolean | If true, runs standard auth validation checks alongside custom auth handlers |
| custom_ui_sso_sign_in_handler | string | Custom handler for SSO sign-in logic in the UI |
| database_connection_pool_timeout | integer | Database connection pool timeout in seconds |
| disable_error_logs | boolean | If true, suppresses error tracking and storage in the database |
| enable_health_check_routing | boolean | If true, enables health check-driven request routing to avoid unhealthy deployments |
| enable_mcp_registry | boolean | If true, enables access to the centralized MCP server registry |
| enforce_rbac | boolean | If true, enables role-based access control (RBAC) for all proxy operations |
| forward_llm_provider_auth_headers | boolean | If true, forwards provider-specific auth headers to LLM API calls |
| health_check_concurrency | integer | Maximum number of concurrent health check operations |
| health_check_staleness_threshold | integer | Maximum age in seconds for health check results before marking deployments as stale |
| maximum_spend_logs_cleanup_cron | string | Cron expression for scheduling automatic spend log cleanup tasks |
| mcp_client_side_auth_header_name | string | HTTP header name for client-side MCP server credentials |
| mcp_internal_ip_ranges | list | CIDR ranges considered internal for non-public MCP server access control |
| mcp_required_fields | list | List of required field names for MCP server submissions |
| mcp_trusted_proxy_ranges | list | CIDR ranges of proxies trusted to forward X-Forwarded-For headers for MCP |
| require_end_user_mcp_access_defined | boolean | If true, requires end users to have explicit MCP access permissions defined |
| role_permissions | list | List of role-based permission configurations |
| search_tools | list | List of search tool configurations for enabling web search capabilities |
| token_rate_limit_type | string | Rate limit counting method: "total", "output", or "input" tokens |
| use_redis_transaction_buffer | boolean | If true, buffers database transactions in Redis before writing |
| use_shared_health_check | boolean | If true, uses Redis-backed shared health check state across multiple proxy instances |
| user_header_mappings | dict | Map custom request headers to user IDs using lookup rules |
| user_header_name | string | HTTP header name to extract user identity from requests |
### router_settings - Reference
@ -367,6 +394,8 @@ router_settings:
| ignore_invalid_deployments | boolean | If true, ignores invalid deployments. Default for proxy is True - to prevent invalid models from blocking other models from being loaded. |
| search_tools | List[SearchToolTypedDict] | List of search tool configurations for Search API integration. Each tool specifies a search_tool_name and litellm_params with search_provider, api_key, api_base, etc. [Further Docs](../search/index.md) |
| guardrail_list | List[GuardrailTypedDict] | List of guardrail configurations for guardrail load balancing. Enables load balancing across multiple guardrail deployments with the same guardrail_name. [Further Docs](./guardrails/guardrail_load_balancing.md) |
| enable_health_check_routing | boolean | If true, enables health check-driven deployment filtering to avoid routing requests to unhealthy deployments |
| health_check_staleness_threshold | integer | Maximum age in seconds for cached health check results before marking deployments as stale |
### environment variables - Reference
@ -804,6 +833,7 @@ router_settings:
| LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS | Optionally enable semantic logs for OTEL
| LITELLM_OTEL_INTEGRATION_ENABLE_METRICS | Optionally enable emantic metrics for OTEL
| LITELLM_ENABLE_PYROSCOPE | If true, enables Pyroscope CPU profiling. Profiles are sent to PYROSCOPE_SERVER_ADDRESS. Off by default. See [Pyroscope profiling](/proxy/pyroscope_profiling).
| LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS | When `true`, if a team's legacy `model_aliases` entry maps a public model name to an internal `model_name_<team_id>_<uuid>` deployment, pre-call handling can skip that rewrite when team-scoped sibling deployments exist for the public name—so load balancing / `order` apply across siblings. Default is `false` for backwards compatibility. See [Team-scoped models and legacy aliases](./load_balancing#team-scoped-models-and-legacy-model_aliases). When stale aliases are detected and this flag is off, the proxy may log a one-time warning.
| PYROSCOPE_APP_NAME | Application name reported to Pyroscope. Required when LITELLM_ENABLE_PYROSCOPE is true. No default.
| PYROSCOPE_SERVER_ADDRESS | Pyroscope server URL to send profiles to. Required when LITELLM_ENABLE_PYROSCOPE is true. No default.
| PYROSCOPE_SAMPLE_RATE | Optional. Sample rate for Pyroscope profiling (integer). No default; when unset, the pyroscope-io library default is used.

View file

@ -314,6 +314,89 @@ general_settings:
health_check_details: False
```
## Health Check Driven Routing
By default, background health checks are observability-only — they populate the `/health` endpoint but don't affect routing. Unhealthy deployments still receive traffic until request failures trigger cooldown.
With `enable_health_check_routing: true`, the router **excludes deployments that failed their last background health check** before selecting a candidate. This gives you proactive failover instead of reactive cooldown.
### How it works
1. Background health checks run on their configured interval
2. After each cycle, every deployment is marked healthy or unhealthy
3. On each incoming request, the router filters out unhealthy deployments **before** cooldown filtering and load balancing
4. If all deployments are unhealthy, the filter is bypassed (safety net — never causes a total outage)
5. If health state is stale (older than `health_check_staleness_threshold`), it is ignored
### Quick start
```yaml
model_list:
- model_name: gpt-4
litellm_params:
model: openai/gpt-4
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-4
litellm_params:
model: openai/gpt-4
api_key: os.environ/OPENAI_API_KEY_SECONDARY
general_settings:
background_health_checks: true
health_check_interval: 60
enable_health_check_routing: true
```
### Configuration
| Setting | Where | Default | Description |
|---------|-------|---------|-------------|
| `enable_health_check_routing` | `general_settings` | `false` | Enable/disable health-check-driven routing |
| `health_check_staleness_threshold` | `general_settings` | `health_check_interval * 2` | Seconds before health state is considered stale and ignored |
| `background_health_checks` | `general_settings` | `false` | Must be `true` for health check routing to work |
| `health_check_interval` | `general_settings` | `300` | Seconds between health check cycles |
### Interaction with cooldown
Health check filtering and cooldown are **additive**. A deployment can be excluded by either mechanism:
- **Health check filter** — proactive, runs on the configured interval, excludes deployments that failed the last check
- **Cooldown** — reactive, triggered by request failures, excludes deployments for a short TTL
This means request failures still provide fast detection between health check intervals.
### Staleness
If a health check result is older than `health_check_staleness_threshold`, it is ignored and the deployment is treated as eligible. This prevents stale data from permanently excluding a deployment if the health check loop stops or slows down.
The default staleness threshold is `health_check_interval * 2`. For a 60s interval, health state expires after 120s.
### Example: custom staleness
```yaml
general_settings:
background_health_checks: true
health_check_interval: 30
enable_health_check_routing: true
health_check_staleness_threshold: 90 # ignore health state older than 90s
```
### Debugging
Run the proxy with `--detailed_debug` and look for:
```
health_check_routing_state_updated healthy=3 unhealthy=1
```
This is logged after each health check cycle when routing state is written.
If the safety net triggers (all deployments unhealthy), you'll see:
```
All deployments marked unhealthy by health checks, bypassing health filter
```
## Health Check Timeout
The health check timeout is set in `litellm/constants.py` and defaults to 60 seconds.

View file

@ -42,9 +42,9 @@ The **High Availability Control Plane** takes a different approach:
<ControlPlaneArchitecture />
The **control plane** is a LiteLLM instance that serves the admin UI and knows about all the workers. It does not proxy LLM requests, it is purely for administration.
The **control plane** is a LiteLLM instance that serves the admin UI and knows about all the workers. It is **not a router** — it does not proxy or route any LLM requests. It exists purely so admins can switch between workers and manage them from a single UI.
Each **worker** is a fully independent LiteLLM proxy that handles LLM requests for its region or team. Workers have their own users, keys, teams, and budgets.
Each **worker** is a fully independent LiteLLM proxy that handles LLM requests for its region or team. Workers have their own database, Redis, users, keys, teams, and budgets. No infrastructure is shared between workers.
## Setup

View file

@ -0,0 +1,318 @@
# JWT → Virtual Key Mapping
:::info Enterprise
JWT → Virtual Key Mapping is an Enterprise feature.
[Get a free trial](https://enterprise.litellm.ai/demo)
:::
Map JWT tokens to LiteLLM virtual keys — so every JWT client gets the same granular controls as a virtual key: model restrictions, spend limits, rate limits, guardrails, and full spend tracking.
**Why this matters:** Standard JWT auth maps a JWT to a *team*. That's a shared boundary — all clients under a team share the same limits. With JWT → Virtual Key Mapping, each individual JWT client (identified by a claim like `client_id`, `azp`, or `sub`) maps to its own virtual key. You get per-client accountability without issuing API keys to your users.
**Common use case:** Your company uses SSO/OIDC. Developers use Claude Code with their identity tokens. You want to enforce per-developer model access and spend limits without giving each person a LiteLLM API key.
---
## How It Works
```mermaid
sequenceDiagram
participant Client as Client (Claude Code / API)
participant Proxy as LiteLLM Proxy
participant OIDC as OIDC Provider
participant DB as Mapping Table
Client->>Proxy: POST /v1/chat/completions<br/>Authorization: Bearer <JWT>
Proxy->>OIDC: Verify JWT signature
OIDC-->>Proxy: Valid ✓
Proxy->>Proxy: Extract claim<br/>(e.g. client_id = "alice@corp.com")
Proxy->>DB: Look up (claim_name, claim_value)
alt Mapping found
DB-->>Proxy: virtual_key_id = sk-abc123
Proxy->>Proxy: Apply virtual key permissions<br/>(models, budget, rate limits)
Proxy-->>Client: 200 OK
else No mapping — fallback_team_mapping
Proxy->>Proxy: Fall through to team JWT auth
Proxy-->>Client: 200 OK
else No mapping — reject
Proxy-->>Client: 403 Forbidden
else No mapping — auto_register
Proxy->>DB: Create new virtual key + mapping
Proxy-->>Client: 200 OK
end
```
---
## Setup
### Prerequisites
Complete [OIDC JWT Auth setup](./token_auth.md) first — you need `JWT_PUBLIC_KEY_URL` configured and `enable_jwt_auth: True` in your proxy config.
### Step 1. Configure the JWT claim to map on
Add `jwt_client_id_field` to your `litellm_jwtauth` config. This is the JWT claim LiteLLM uses as the lookup key:
```yaml
general_settings:
master_key: sk-1234
enable_jwt_auth: True
litellm_jwtauth:
team_id_jwt_field: "team_id" # existing team mapping (optional)
user_id_jwt_field: "sub"
jwt_client_id_field: "client_id" # 👈 claim used for key mapping
unregistered_jwt_client_behavior: "fallback_team_mapping" # see below
```
**`unregistered_jwt_client_behavior`** controls what happens when a JWT has no registered mapping:
| Value | Behavior |
|-------|----------|
| `fallback_team_mapping` | Fall through to team-based JWT auth (default — backward compatible) |
| `reject` | Return 403 if no mapping found |
| `auto_register` | Auto-create a virtual key + mapping on first encounter |
### Step 2. Register a JWT client → virtual key mapping
**Option A: Single call (creates key + mapping atomically)**
```bash
curl -X POST 'http://0.0.0.0:4000/jwt_client/new' \
-H 'Authorization: Bearer <PROXY_MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "dev-alice",
"models": ["claude-sonnet-4-5", "claude-haiku-4-5"],
"max_budget": 50.0,
"budget_duration": "30d",
"rpm_limit": 100,
"tpm_limit": 50000,
"team_id": "engineering"
}'
```
Response includes the virtual key token (only shown on creation):
```json
{
"key": "sk-abc123...",
"key_id": "key_123",
"mapping_id": "mapping_456",
"jwt_claim_name": "client_id",
"jwt_claim_value": "dev-alice"
}
```
**Option B: Map an existing virtual key**
```bash
curl -X POST 'http://0.0.0.0:4000/jwt/key/mapping/new' \
-H 'Authorization: Bearer <PROXY_MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "dev-alice",
"virtual_key_id": "key_123"
}'
```
### Step 3. Test it
```bash
# Get a JWT from your OIDC provider (must have client_id: dev-alice)
JWT_TOKEN="eyJhbG..."
curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
-H "Authorization: Bearer $JWT_TOKEN" \
-H 'Content-Type: application/json' \
-d '{
"model": "claude-sonnet-4-5",
"messages": [{"role": "user", "content": "Hello"}]
}'
```
The request is now tracked against `dev-alice`'s virtual key — spend, rate limits, and model access enforced per-client.
---
## Walkthrough: Admin grants granular access, team uses Claude Code
This is the full flow for an engineering team using Claude Code with company SSO.
### Admin setup
**1. Create a team for engineering**
```bash
curl -X POST 'http://0.0.0.0:4000/team/new' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"team_alias": "engineering",
"models": ["claude-sonnet-4-5", "claude-haiku-4-5"]
}'
```
**2. Register each developer with their own key and spend limit**
```bash
# Alice — senior eng, higher budget
curl -X POST 'http://0.0.0.0:4000/jwt_client/new' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "alice@corp.com",
"team_id": "engineering",
"models": ["claude-sonnet-4-5", "claude-haiku-4-5"],
"max_budget": 200.0,
"budget_duration": "30d",
"rpm_limit": 200
}'
# Bob — contractor, tighter limits
curl -X POST 'http://0.0.0.0:4000/jwt_client/new' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "bob@contractor.com",
"team_id": "engineering",
"models": ["claude-haiku-4-5"],
"max_budget": 20.0,
"budget_duration": "30d",
"rpm_limit": 30
}'
```
**3. Configure Claude Code to use the proxy**
Set the proxy as the API base in your team's Claude Code config:
```bash
# Point Claude Code at the LiteLLM proxy instead of Anthropic directly.
# ANTHROPIC_API_KEY here is the bearer token sent to the proxy — set it to
# the user's SSO/OIDC JWT token (obtained from your IdP at login).
export ANTHROPIC_API_KEY="<user-sso-jwt-token>"
export ANTHROPIC_BASE_URL="http://your-litellm-proxy:4000"
```
Or in `~/.claude/settings.json`:
```json
{
"env": {
"ANTHROPIC_BASE_URL": "http://your-litellm-proxy:4000"
}
}
```
**4. Developers authenticate with SSO as usual**
When Alice runs Claude Code, her JWT (issued by your IdP with `client_id: alice@corp.com`) goes to the proxy. LiteLLM looks up the mapping, finds her virtual key, and enforces her specific limits — her $200/month budget, 200 RPM cap, and access to Sonnet and Haiku only.
Bob's token maps to his own key — $20/month, Haiku only, 30 RPM.
No API keys distributed. No shared limits. Full per-developer spend visibility in the LiteLLM dashboard.
---
## Managing mappings
**View a mapping + its key settings**
```bash
curl 'http://0.0.0.0:4000/jwt/key/mapping/info?jwt_claim_name=client_id&jwt_claim_value=alice@corp.com' \
-H 'Authorization: Bearer <MASTER_KEY>'
```
Response includes the linked key's `models`, `max_budget`, `spend`, `rpm_limit`, `expires`, etc.
**Update a mapping**
```bash
curl -X POST 'http://0.0.0.0:4000/jwt_client/update' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "alice@corp.com",
"max_budget": 300.0
}'
```
**Delete a mapping**
```bash
curl -X DELETE 'http://0.0.0.0:4000/jwt/key/mapping/delete' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "client_id",
"jwt_claim_value": "alice@corp.com"
}'
```
---
## Security
JWT-bound keys are locked down:
- Non-admin users cannot call `/key/update`, `/key/delete`, or `/key/regenerate` on a JWT-bound key. These return 403.
- JWT-bound keys are automatically restricted to `llm_api_routes` — they can make LLM calls but cannot manage other keys or admin resources.
- Only proxy admins can create, update, or delete mappings.
---
## Multi-IdP support
If you have users across multiple identity providers that share the same claim values (e.g. two services both have `sub: user-123` from different issuers), set `issuer` when creating the mapping:
```bash
curl -X POST 'http://0.0.0.0:4000/jwt_client/new' \
-H 'Authorization: Bearer <MASTER_KEY>' \
-H 'Content-Type: application/json' \
-d '{
"jwt_claim_name": "sub",
"jwt_claim_value": "user-123",
"issuer": "https://idp-a.corp.com",
"models": ["claude-sonnet-4-5"],
"max_budget": 50.0
}'
```
Mappings are unique per `(claim_name, claim_value, issuer)` — so `user-123` from IdP A and `user-123` from IdP B resolve to different virtual keys.
---
## What JWT clients can and can't do vs virtual keys
| Capability | Virtual Key | JWT → Key Mapping |
|---|---|---|
| Per-client model access | ✅ | ✅ |
| Per-client spend budget | ✅ | ✅ |
| Per-client RPM/TPM limits | ✅ | ✅ |
| Team membership | ✅ | ✅ |
| Spend tracking in dashboard | ✅ | ✅ |
| Guardrails | ✅ | ✅ |
| Key rotation | ✅ | ✅ (admin only) |
| Key expiry | ✅ | ✅ |
| No API key distribution needed | ❌ | ✅ |
| Works with existing SSO/OIDC | ❌ | ✅ |
---
## Related
- [OIDC JWT Auth](./token_auth.md) — base JWT auth setup required before using this feature
- [Virtual Keys](./virtual_keys.md) — full virtual key documentation
- [Access Control](./access_control.md) — model and team access control

View file

@ -324,17 +324,58 @@ model_list:
litellm_params:
model: azure/gpt-4-fallback
api_key: os.environ/AZURE_API_KEY_2
order: 2 # 👈 Used when order=1 is unavailable
router_settings:
enable_pre_call_checks: true # 👈 Required for 'order' to work
order: 2 # 👈 Used when order=1 fails
```
:::important
The `order` parameter requires `enable_pre_call_checks: true` in `router_settings`.
:::
### How order-based fallback works
If `order=1` deployment is unavailable (e.g., rate-limited), the router falls back to `order=2` deployments.
When a request to an `order=1` deployment fails (connection error, 404, 429, etc.), the router automatically tries `order=2` deployments, then `order=3`, and so on. Each order level gets its own set of retries before escalating to the next.
If all order levels are exhausted, the router falls through to any configured [model-level fallbacks](#fallbacks).
```yaml
model_list:
- model_name: gpt-4
litellm_params:
model: azure/gpt-4-primary
api_key: os.environ/AZURE_API_KEY
order: 1
- model_name: gpt-4
litellm_params:
model: azure/gpt-4-secondary
api_key: os.environ/AZURE_API_KEY_2
order: 2
- model_name: gpt-4-fallback
litellm_params:
model: openai/gpt-4
api_key: os.environ/OPENAI_API_KEY
router_settings:
fallbacks:
- gpt-4:
- gpt-4-fallback # tried after all order levels fail
```
The fallback chain for the above config: `order=1` → `order=2` → `gpt-4-fallback`.
For 429 (rate limit) errors specifically, the failed deployment is immediately placed on cooldown. If all `order=1` deployments are on cooldown, the router picks `order=2` deployments directly during retries without waiting for the fallback path.
### Team-scoped models and legacy `model_aliases` {#team-scoped-models-and-legacy-model_aliases}
Team-scoped deployments are identified by `model_info.team_id` and `model_info.team_public_model_name`. Requests should use the **public** model name; the router resolves all sibling deployments (same public name, different `api_base` / `order`, etc.) for routing, failover, and deployment `order`.
For router internals: when a `team_id` is in scope, optimized lookups key off `(team_id, team_public_model_name)`. If code passes an internal deployment id (e.g. `model_name_<team_id>_<uuid>`) instead of the public name, routing still works via the usual deployment-name paths, but the team-specific fast path applies only to the public name.
**Legacy teams:** Older proxy versions could persist `model_aliases` on the team row mapping a public name to a single internal deployment id (`model_name_<team_id>_<uuid>`). On each request, pre-call logic may still rewrite `model` to that internal name **before** routing, which collapses to one deployment and can make newer sibling deployments unreachable.
**Migration options:**
1. **Recommended for upgrades:** Set environment variable `LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS=true` so that when sibling team deployments exist for the public name, the stale alias rewrite is skipped and team-scoped routing (including `order` and failover) applies. See the [Environment variables](./config_settings) table in the proxy settings doc.
2. **Data cleanup:** Remove obsolete `model_aliases` entries for team public names from the team record in the database so only `team_public_model_name` + team model list drive access.
If a stale alias is detected and the bypass is **not** enabled, the proxy may emit a **one-time** warning in logs explaining that sibling deployments may be unreachable until the flag is set or aliases are cleaned up.
### When You'll See Load Balancing in Action

View file

@ -16,6 +16,12 @@ Use JWT's to auth admins / users / projects into the proxy.
:::
:::tip JWT → Virtual Key Mapping
Want per-user model restrictions, spend limits, and rate limits without distributing API keys? See **[JWT → Virtual Key Mapping](./jwt_key_mapping.md)** — enterprise-grade granular access control for JWT-authenticated users (e.g. Claude Code + SSO).
:::
## Usage
### Step 1. Setup Proxy

View file

@ -842,6 +842,8 @@ Traffic mirroring allows you to "mimic" production traffic to a secondary (silen
Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them.
When a request to an `order=1` deployment fails (connection error, 404, 429, etc.), the router automatically tries `order=2` deployments, then `order=3`, and so on. Each order level gets its own set of retries before escalating to the next. If all order levels are exhausted, the router falls through to any configured [fallbacks](#fallbacks).
<Tabs>
<TabItem value="sdk" label="SDK">
@ -862,18 +864,14 @@ model_list = [
"litellm_params": {
"model": "azure/gpt-4-fallback",
"api_key": os.getenv("AZURE_API_KEY_2"),
"order": 2, # 👈 Used when order=1 is unavailable
"order": 2, # 👈 Tried when order=1 fails
},
},
]
router = Router(model_list=model_list, enable_pre_call_checks=True) # 👈 Required for 'order' to work
router = Router(model_list=model_list)
```
:::important
The `order` parameter requires `enable_pre_call_checks=True` to be set on the Router.
:::
</TabItem>
<TabItem value="proxy" label="PROXY">
@ -889,10 +887,7 @@ model_list:
litellm_params:
model: azure/gpt-4-fallback
api_key: os.environ/AZURE_API_KEY_2
order: 2 # 👈 Used when order=1 is unavailable
router_settings:
enable_pre_call_checks: true # 👈 Required for 'order' to work
order: 2 # 👈 Tried when order=1 fails
```
</TabItem>

View file

@ -285,6 +285,7 @@ const config = {
to: "docs/enterprise"
},
{ to: '/blog', label: 'Blog', position: 'left' },
{ to: '/release_notes', label: 'Release Notes', position: 'left' },
{
href: 'https://github.com/BerriAI/litellm',
position: 'right',

Binary file not shown.

After

Width:  |  Height:  |  Size: 158 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 16 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 52 KiB

View file

@ -48,27 +48,26 @@
"node": ">=16.14",
"npm": ">=8.3.0"
},
"resolutions": {
"webpack-dev-server": ">=5.2.1",
"form-data": ">=4.0.4",
"mermaid": ">=11.10.0",
"gray-matter": "4.0.3",
"node-forge": ">=1.3.2"
},
"overrides": {
"webpack-dev-server": ">=5.2.1",
"form-data": ">=4.0.4",
"mermaid": ">=11.10.0",
"gray-matter": "4.0.3",
"glob": ">=11.1.0",
"tar": ">=7.5.10",
"minimatch": ">=10.2.4",
"diff": ">=8.0.3",
"@isaacs/brace-expansion": ">=5.0.1",
"serialize-javascript": ">=7.0.3",
"node-forge": ">=1.3.2",
"mdast-util-to-hast": ">=13.2.1",
"lodash-es": ">=4.17.23",
"webpack-dev-server": "5.2.3",
"form-data": "4.0.5",
"mermaid": "11.12.1",
"minimatch": "10.2.4",
"serialize-javascript": "7.0.3",
"mdast-util-to-hast": "13.2.1",
"lodash-es": "4.17.23",
"@babel/traverse": "7.28.5",
"ws": "8.19.0",
"http-proxy-middleware": "3.0.5",
"tar-fs": "3.1.1",
"webpack-dev-middleware": "5.3.4",
"braces": "3.0.3",
"webpack": "5.105.3",
"serve-static": "2.2.1",
"path-to-regexp": "1.9.0",
"dompurify": "3.3.2",
"svgo": "4.0.1",
"schema-utils@3": {
"ajv": "6.14.0"
},
@ -83,18 +82,6 @@
},
"url-loader": {
"ajv": "6.14.0"
},
"@babel/traverse": ">=7.23.2",
"ws": ">=7.5.10",
"http-proxy-middleware": ">=2.0.9",
"tar-fs": ">=2.1.4",
"webpack-dev-middleware": ">=5.3.4",
"braces": ">=3.0.3",
"axios": ">=0.30.2",
"webpack": ">=5.94.0",
"serve-static": ">=1.16.0",
"path-to-regexp": ">=0.1.12",
"dompurify": ">=3.3.2",
"svgo": ">=3.3.3"
}
}
}

View file

@ -0,0 +1,62 @@
---
title: "v1.83.0 - Official Release (Post Supply Chain Incident)"
slug: "v1-83-0"
date: 2026-03-31T00:00:00
authors:
- name: Krrish Dholakia
title: CEO, LiteLLM
url: https://www.linkedin.com/in/krish-d/
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
- name: Ishaan Jaff
title: CTO, LiteLLM
url: https://www.linkedin.com/in/reffajnaahsi/
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
hide_table_of_contents: false
---
## Deploy this version
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
<Tabs>
<TabItem value="docker" label="Docker">
``` showLineNumbers title="docker run litellm"
docker run \
-e STORE_MODEL_IN_DB=True \
-p 4000:4000 \
ghcr.io/berriai/litellm:main-1.83.0-nightly
```
</TabItem>
<TabItem value="pip" label="Pip">
``` showLineNumbers title="pip install litellm"
pip install litellm==1.83.0
```
</TabItem>
</Tabs>
## Context: First Release After Supply Chain Incident
v1.83.0 is the first LiteLLM release built and published through our new [CI/CD v2 pipeline](https://docs.litellm.ai/blog/ci-cd-v2-improvements), following the [supply chain incident on March 24](https://docs.litellm.ai/blog/security-update-march-2026).
We paused all releases for one week while we:
1. Completed a forensic review with [Mandiant](https://www.mandiant.com/) and [Veria Labs](https://verialabs.com/)
2. Rebuilt the release pipeline from scratch with isolated environments and ephemeral credentials
3. Verified the codebase contains no indicators of compromise
If you have questions about this release or the incident, see our [Security Townhall post](https://docs.litellm.ai/blog/security-townhall-updates) or reach out at `security@berri.ai`.
---
## Links
- **PyPI**: [litellm 1.83.0](https://pypi.org/project/litellm/1.83.0/)
- **Security update**: [Supply chain incident report](https://docs.litellm.ai/blog/security-update-march-2026)
- **Security townhall**: [What happened, what we've done, what comes next](https://docs.litellm.ai/blog/security-townhall-updates)
- **CI/CD v2**: [Announcing CI/CD v2 for LiteLLM](https://docs.litellm.ai/blog/ci-cd-v2-improvements)
- **April stability sprint**: [Help us plan](https://github.com/BerriAI/litellm/issues/24825)

View file

@ -452,6 +452,7 @@ const sidebars = {
items: [
"proxy/virtual_keys",
"proxy/token_auth",
"proxy/jwt_key_mapping",
"proxy/service_accounts",
"proxy/access_control",
"proxy/cli_sso",
@ -850,6 +851,7 @@ const sidebars = {
items: [
"providers/gemini",
"providers/gemini/videos",
"providers/gemini/music",
"providers/google_ai_studio/files",
"providers/google_ai_studio/image_gen",
"providers/google_ai_studio/realtime",

View file

@ -3,23 +3,56 @@ import styles from './styles.module.css';
/* ────────────────────── Shared small pieces ────────────────────── */
function InfraChip({ color, label }: { color: string; label: string }) {
const dotClass =
function InfraBox({ icon, label, color }: { icon: string; label: string; color: 'green' | 'blue' | 'orange' }) {
const colorClass =
color === 'green'
? styles.infraDotGreen
? styles.infraBoxGreen
: color === 'blue'
? styles.infraDotBlue
: styles.infraDotOrange;
? styles.infraBoxBlue
: styles.infraBoxOrange;
return (
<span className={styles.infraChip}>
<span className={`${styles.infraDot} ${dotClass}`} />
{label}
</span>
<div className={`${styles.infraBox} ${colorClass}`}>
<span className={styles.infraBoxIcon}>{icon}</span>
<span className={styles.infraBoxLabel}>{label}</span>
</div>
);
}
/* ────────────────────── Architecture tab ────────────────────── */
/* ────────────────────── Worker column with infra ────────────────────── */
function WorkerColumn({
name,
region,
subtitle,
nodeClass,
badgeClass,
}: {
name: string;
region: string;
subtitle: string;
nodeClass: string;
badgeClass: string;
}) {
return (
<div className={styles.workerColumn}>
<div className={`${styles.node} ${styles.nodeWorker} ${nodeClass}`}>
<div className={styles.nodeHeader}>
<span className={styles.nodeTitle}>{name}</span>
<span className={`${styles.badge} ${badgeClass}`}>{region}</span>
</div>
<div className={styles.nodeSubtitle}>{subtitle}</div>
<div className={styles.nodeCaption}>Handles LLM requests</div>
</div>
<div className={styles.infraStack}>
<InfraBox icon="🗄" label="Own Database" color="green" />
<InfraBox icon="⚡" label="Own Redis" color="orange" />
</div>
</div>
);
}
/* ────────────────────── Architecture diagram ────────────────────── */
function ArchitectureView() {
return (
@ -36,49 +69,41 @@ function ArchitectureView() {
<div className={`${styles.node} ${styles.nodeControlPlane}`}>
<div className={styles.nodeHeader}>
<span className={styles.nodeTitle}>Control Plane</span>
<span className={`${styles.badge} ${styles.badgeBlue}`}>UI</span>
<span className={`${styles.badge} ${styles.badgeBlue}`}>ADMIN UI ONLY</span>
</div>
<div className={styles.nodeSubtitle}>cp.example.com</div>
<div className={styles.infraRow}>
<InfraChip color="green" label="Own DB" />
<InfraChip color="orange" label="Own Redis" />
<InfraChip color="blue" label="Own Key" />
<div className={styles.nodeCaption}>
Not a router — does not proxy LLM requests.
<br />
Lets admins switch between workers to manage them.
</div>
</div>
{/* Branch connector */}
<div className={styles.connectorBranch}>
<div className={`${styles.branchLeg} ${styles.branchLegLeft}`} />
<div className={`${styles.branchLeg} ${styles.branchLegRight}`} />
{/* Branch connector with label */}
<div className={styles.connectorBranchLabeled}>
<span className={styles.connectorLabel}>UI management only</span>
<div className={styles.connectorBranch}>
<div className={`${styles.branchLeg} ${styles.branchLegLeft}`} />
<div className={`${styles.branchLeg} ${styles.branchLegRight}`} />
</div>
</div>
{/* Workers */}
<div className={styles.workersRow}>
<div className={`${styles.node} ${styles.nodeWorker} ${styles.nodeWorkerA}`}>
<div className={styles.nodeHeader}>
<span className={styles.nodeTitle}>Worker A</span>
<span className={`${styles.badge} ${styles.badgeGreen}`}>US East</span>
</div>
<div className={styles.nodeSubtitle}>worker-a.example.com</div>
<div className={styles.infraRow}>
<InfraChip color="green" label="Own DB" />
<InfraChip color="orange" label="Own Redis" />
<InfraChip color="blue" label="Own Key" />
</div>
</div>
<div className={`${styles.node} ${styles.nodeWorker} ${styles.nodeWorkerB}`}>
<div className={styles.nodeHeader}>
<span className={styles.nodeTitle}>Worker B</span>
<span className={`${styles.badge} ${styles.badgePurple}`}>EU West</span>
</div>
<div className={styles.nodeSubtitle}>worker-b.example.com</div>
<div className={styles.infraRow}>
<InfraChip color="green" label="Own DB" />
<InfraChip color="orange" label="Own Redis" />
<InfraChip color="blue" label="Own Key" />
</div>
</div>
<WorkerColumn
name="Worker A"
region="US East"
subtitle="worker-a.example.com"
nodeClass={styles.nodeWorkerA}
badgeClass={styles.badgeGreen}
/>
<WorkerColumn
name="Worker B"
region="EU West"
subtitle="worker-b.example.com"
nodeClass={styles.nodeWorkerB}
badgeClass={styles.badgePurple}
/>
</div>
</div>
);

View file

@ -284,45 +284,67 @@
color: var(--cp-purple);
}
/* ── Infrastructure chips ── */
.infraRow {
display: flex;
gap: 0.4rem;
justify-content: center;
flex-wrap: wrap;
margin-top: 0.5rem;
/* ── Node caption ── */
.nodeCaption {
font-size: 0.72rem;
color: var(--cp-text-muted);
margin-top: 0.4rem;
line-height: 1.4;
font-style: italic;
}
.infraChip {
/* ── Infrastructure boxes (per-worker) ── */
.infraStack {
display: flex;
flex-direction: column;
gap: 0.35rem;
margin-top: 0.5rem;
width: 100%;
}
.infraBox {
display: flex;
align-items: center;
gap: 0.3rem;
font-size: 0.7rem;
font-weight: 500;
color: var(--cp-text-secondary);
background: var(--cp-infra-bg);
border: 1px solid var(--cp-infra-border);
border-radius: 6px;
padding: 0.2rem 0.5rem;
gap: 0.5rem;
padding: 0.45rem 0.75rem;
border-radius: 8px;
border: 1.5px solid var(--cp-border);
background: var(--cp-card-bg);
}
.infraDot {
width: 6px;
height: 6px;
border-radius: 50%;
.infraBoxGreen {
border-color: var(--cp-green);
background: var(--cp-green-light);
}
.infraBoxBlue {
border-color: var(--cp-accent);
background: var(--cp-accent-light);
}
.infraBoxOrange {
border-color: var(--cp-orange);
background: var(--cp-orange-light);
}
.infraBoxIcon {
font-size: 0.85rem;
flex-shrink: 0;
}
.infraDotGreen {
background: var(--cp-green);
.infraBoxLabel {
font-size: 0.75rem;
font-weight: 600;
color: var(--cp-text);
}
.infraDotBlue {
background: var(--cp-accent);
}
.infraDotOrange {
background: var(--cp-orange);
/* ── Worker column (card + infra stack) ── */
.workerColumn {
display: flex;
flex-direction: column;
align-items: stretch;
min-width: 220px;
max-width: 260px;
}
/* ── Workers row ── */
@ -333,6 +355,24 @@
flex-wrap: wrap;
}
/* ── Connector with label ── */
.connectorBranchLabeled {
display: flex;
flex-direction: column;
align-items: center;
width: 100%;
max-width: 700px;
}
.connectorLabel {
font-size: 0.7rem;
color: var(--cp-text-muted);
font-weight: 500;
text-transform: uppercase;
letter-spacing: 0.05em;
margin-bottom: 0.25rem;
}
/* ── Animated flow ── */
.flowLabel {
font-size: 0.7rem;
@ -511,6 +551,16 @@
max-width: 260px;
}
.workerColumn {
min-width: auto;
width: 100%;
max-width: 280px;
}
.connectorBranchLabeled {
display: none;
}
.connectorBranch {
display: none;
}

View file

@ -0,0 +1,84 @@
import React, { useState } from "react";
import styles from "./styles.module.css";
interface VersionEntry {
version: string;
sha256: string;
gitCommit: string;
}
interface Props {
entries: VersionEntry[];
}
function CopyButton({ text }: { text: string }) {
const [copied, setCopied] = useState(false);
const handleCopy = () => {
navigator.clipboard.writeText(text).then(() => {
setCopied(true);
setTimeout(() => setCopied(false), 1500);
});
};
return (
<button
className={styles.copyBtn}
onClick={handleCopy}
title="Copy full SHA-256"
>
{copied ? "✓" : "⧉"}
</button>
);
}
export default function VersionVerificationTable({ entries }: Props) {
return (
<div className={styles.wrapper}>
<table className={styles.table}>
<thead>
<tr>
<th>Version</th>
<th>SHA-256</th>
<th>Clean of IOCs</th>
<th>Matches Git</th>
<th>Git Commit</th>
<th>Status</th>
</tr>
</thead>
<tbody>
{entries.map((entry) => (
<tr key={`${entry.version}-${entry.gitCommit}`}>
<td className={styles.version}>{entry.version}</td>
<td>
<span className={styles.sha}>
<code>{entry.sha256.slice(0, 16)}…</code>
<CopyButton text={entry.sha256} />
</span>
</td>
<td>
<span className={styles.badgeClean}>✔ CLEAN</span>
</td>
<td>
<span className={styles.badgeYes}>✔ YES</span>
</td>
<td>
<a
className={styles.commitLink}
href={`https://github.com/BerriAI/litellm/commit/${entry.gitCommit}`}
target="_blank"
rel="noopener noreferrer"
>
{entry.gitCommit}
</a>
</td>
<td>
<span className={styles.badgeClean}>✔ CLEAN</span>
</td>
</tr>
))}
</tbody>
</table>
</div>
);
}

View file

@ -0,0 +1,106 @@
.wrapper {
overflow-x: auto;
margin: 1rem 0;
}
.table {
width: 100%;
border-collapse: separate;
border-spacing: 0;
font-size: 0.9rem;
border: 1px solid var(--ifm-color-emphasis-300);
border-radius: 8px;
overflow: hidden;
}
.table th,
.table td {
padding: 0.6rem 0.75rem;
text-align: left;
white-space: nowrap;
}
.table thead th {
background: var(--ifm-color-emphasis-200);
font-weight: 600;
font-size: 0.8rem;
text-transform: uppercase;
letter-spacing: 0.03em;
color: var(--ifm-color-emphasis-700);
border-bottom: 2px solid var(--ifm-color-emphasis-300);
}
.table tbody tr:nth-child(even) {
background: var(--ifm-color-emphasis-100);
}
.table tbody tr:hover {
background: var(--ifm-color-emphasis-200);
}
.table tbody td {
border-bottom: 1px solid var(--ifm-color-emphasis-200);
}
.table tbody tr:last-child td {
border-bottom: none;
}
.badge {
display: inline-flex;
align-items: center;
gap: 4px;
padding: 2px 8px;
border-radius: 12px;
font-size: 0.75rem;
font-weight: 600;
line-height: 1.4;
}
.badgeClean {
composes: badge;
background: #d4edda;
color: #155724;
}
.badgeYes {
composes: badge;
background: #cce5ff;
color: #004085;
}
.sha {
display: inline-flex;
align-items: center;
gap: 4px;
font-family: var(--ifm-font-family-monospace);
font-size: 0.8rem;
}
.copyBtn {
display: inline-flex;
align-items: center;
justify-content: center;
background: none;
border: 1px solid var(--ifm-color-emphasis-300);
border-radius: 4px;
cursor: pointer;
padding: 2px 4px;
font-size: 0.7rem;
color: var(--ifm-color-emphasis-600);
transition: background 0.15s, color 0.15s;
}
.copyBtn:hover {
background: var(--ifm-color-emphasis-200);
color: var(--ifm-color-emphasis-800);
}
.commitLink {
font-family: var(--ifm-font-family-monospace);
font-size: 0.8rem;
}
.version {
font-weight: 600;
}

Binary file not shown.

After

Width:  |  Height:  |  Size: 39 KiB

View file

@ -116,6 +116,7 @@ class PagerDutyAlerting(SlackAlerting):
user_api_key_org_id=_meta.get("user_api_key_org_id"),
user_api_key_team_id=_meta.get("user_api_key_team_id"),
user_api_key_project_id=_meta.get("user_api_key_project_id"),
user_api_key_project_alias=_meta.get("user_api_key_project_alias"),
user_api_key_user_id=_meta.get("user_api_key_user_id"),
user_api_key_team_alias=_meta.get("user_api_key_team_alias"),
user_api_key_end_user_id=_meta.get("user_api_key_end_user_id"),
@ -197,6 +198,7 @@ class PagerDutyAlerting(SlackAlerting):
user_api_key_org_id=user_api_key_dict.org_id,
user_api_key_team_id=user_api_key_dict.team_id,
user_api_key_project_id=user_api_key_dict.project_id,
user_api_key_project_alias=user_api_key_dict.project_alias,
user_api_key_user_id=user_api_key_dict.user_id,
user_api_key_team_alias=user_api_key_dict.team_alias,
user_api_key_end_user_id=user_api_key_dict.end_user_id,

View file

@ -3,7 +3,7 @@ Polls LiteLLM_ManagedObjectTable to check if the batch job is complete, and if t
"""
from datetime import datetime, timedelta, timezone
from typing import TYPE_CHECKING, Optional
from typing import TYPE_CHECKING, List, Optional, Tuple
from litellm._logging import verbose_proxy_logger
from litellm._uuid import uuid
@ -131,6 +131,15 @@ class CheckBatchCost:
get_model_id_from_unified_batch_id,
)
try:
from litellm.integrations.prometheus import PrometheusLogger
prom_logger = PrometheusLogger.get_instance()
except Exception as e:
verbose_proxy_logger.error(f"CheckBatchCost: could not get Prometheus logger: {e}")
prom_logger = None
processed_models: List[Tuple[Optional[str], Optional[str]]] = []
try:
await self._cleanup_stale_managed_objects()
except Exception as cleanup_err:
@ -189,6 +198,8 @@ class CheckBatchCost:
verbose_proxy_logger.info(
f"Skipping job {unified_object_id} because it is not a valid unified object id"
)
if prom_logger:
prom_logger.record_check_batch_cost_error("invalid_unified_id")
continue
else:
unified_object_id = decoded_unified_object_id
@ -200,6 +211,8 @@ class CheckBatchCost:
verbose_proxy_logger.info(
f"Skipping job {unified_object_id} because it is not a valid model id"
)
if prom_logger:
prom_logger.record_check_batch_cost_error("invalid_model_id")
continue
verbose_proxy_logger.info(
@ -219,6 +232,8 @@ class CheckBatchCost:
verbose_proxy_logger.info(
f"Skipping job {unified_object_id} because of error querying model ID: {model_id} for cost and usage of batch ID: {batch_id}: {e}"
)
if prom_logger:
prom_logger.record_check_batch_cost_error("provider_retrieval_error")
continue
## RETRIEVE THE BATCH JOB OUTPUT FILE
@ -276,11 +291,25 @@ class CheckBatchCost:
content_bytes # type: ignore[arg-type]
)
# Record output file size
if prom_logger and content_bytes:
try:
prom_logger.record_managed_file_size(
size_bytes=len(content_bytes), # type: ignore
purpose="batch",
file_type="output",
model=model_id,
)
except Exception:
pass
deployment_info = self.llm_router.get_deployment(model_id=model_id)
if deployment_info is None:
verbose_proxy_logger.info(
f"Skipping job {unified_object_id} because it is not a valid deployment info"
)
if prom_logger:
prom_logger.record_check_batch_cost_error("deployment_not_found")
continue
custom_llm_provider = deployment_info.litellm_params.custom_llm_provider
litellm_model_name = deployment_info.litellm_params.model
@ -343,6 +372,19 @@ class CheckBatchCost:
batch_models=batch_models,
)
# Record batch duration (completed_at - created_at)
if prom_logger and response.completed_at and response.created_at:
duration_seconds = float(response.completed_at - response.created_at)
if duration_seconds >= 0:
prom_logger.record_managed_batch_duration(
duration_seconds=duration_seconds,
model=model_name,
api_provider=str(llm_provider) if llm_provider else None,
)
# Track this job for the final metrics summary
processed_models.append((model_name, str(llm_provider) if llm_provider else None))
# mark the job as complete
try:
update_data: dict = {
@ -359,3 +401,10 @@ class CheckBatchCost:
verbose_proxy_logger.error(
f"CheckBatchCost: failed to mark job {job.id} complete in DB: {db_err}"
)
# Record polling run metrics (always, even if nothing was processed)
if prom_logger:
prom_logger.record_check_batch_cost_run(
jobs_polled=len(jobs),
processed_models=processed_models if processed_models else None,
)

View file

@ -74,6 +74,13 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
self.internal_usage_cache = internal_usage_cache
self.prisma_client = prisma_client
@staticmethod
def _get_prometheus_logger():
"""Find PrometheusLogger from litellm.callbacks, if registered."""
from litellm.integrations.prometheus import PrometheusLogger
return PrometheusLogger.get_instance()
async def store_unified_file_id(
self,
file_id: str,
@ -919,6 +926,31 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
model_mappings=model_mappings,
user_api_key_dict=user_api_key_dict,
)
# Emit Prometheus metrics for managed file creation
prom_logger = self._get_prometheus_logger()
if prom_logger:
first_model = target_model_names_list[0] if target_model_names_list else None
first_provider = ""
if responses:
first_provider = getattr(responses[0], "_hidden_params", {}).get("custom_llm_provider") or ""
prom_logger.record_managed_file_created(
model=first_model or "",
api_provider=first_provider,
user=user_api_key_dict.user_id or "",
user_email=getattr(user_api_key_dict, "user_email", None) or "",
api_key_alias=user_api_key_dict.key_alias or "",
)
if response.bytes and response.bytes > 0:
prom_logger.record_managed_file_size(
size_bytes=response.bytes,
purpose=response.purpose or "batch",
file_type="input",
model=first_model,
api_provider=first_provider,
user=user_api_key_dict.user_id,
)
return response
@staticmethod
@ -1105,6 +1137,31 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
file_purpose="batch",
user_api_key_dict=user_api_key_dict,
)
# Only record batch creation metric on actual create (not retrieve/cancel).
# unified_file_id in _hidden_params is only set by the create_batch endpoint.
original_unified_file_id = response._hidden_params.get("unified_file_id")
if original_unified_file_id:
prom_logger = self._get_prometheus_logger()
if prom_logger:
batch_provider = ""
if model_name:
try:
from litellm.litellm_core_utils.get_llm_provider_logic import (
get_llm_provider,
)
_, batch_provider, _, _ = get_llm_provider(model=model_name)
except Exception:
if "/" in model_name:
batch_provider = model_name.split("/")[0]
prom_logger.record_managed_batch_created(
model=model_name or "",
api_provider=batch_provider,
user=user_api_key_dict.user_id or "",
user_email=getattr(user_api_key_dict, "user_email", None) or "",
api_key_alias=user_api_key_dict.key_alias or "",
)
elif isinstance(response, LiteLLMFineTuningJob):
## Check if unified_file_id is in the response
unified_file_id = response._hidden_params.get(
@ -1373,6 +1430,11 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
f"To delete this file before complete cost tracking, please delete or cancel the referencing batch(es) first. "
f"Alternatively, wait for all batches to complete and for cost to be computed (batch_processed=true)."
)
# Record blocked deletion metric
prom_logger = self._get_prometheus_logger()
if prom_logger:
prom_logger.record_managed_file_deleted(result="blocked")
raise HTTPException(
status_code=400,
@ -1408,6 +1470,12 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
file_id, litellm_parent_otel_span
)
# Record successful deletion metric only on actual success
if stored_file_object or delete_response:
prom_logger = self._get_prometheus_logger()
if prom_logger:
prom_logger.record_managed_file_deleted(result="success")
if stored_file_object:
return stored_file_object
elif delete_response:

5
litellm-js/proxy/.npmrc Normal file
View file

@ -0,0 +1,5 @@
# Supply-chain hardening
# Packages needing lifecycle scripts: npm rebuild <pkg>
ignore-scripts=true
# Protects local npm install only — npm ci (used in CI) ignores this
min-release-age=3d

View file

@ -0,0 +1,5 @@
# Supply-chain hardening
# Packages needing lifecycle scripts: npm rebuild <pkg>
ignore-scripts=true
# Protects local npm install only — npm ci (used in CI) ignores this
min-release-age=3d

View file

@ -8,7 +8,7 @@ WORKDIR /app
COPY ./litellm-js/spend-logs/package*.json ./
# Install dependencies
RUN npm install
RUN npm ci
# Install Prisma globally
RUN npm install -g prisma

View file

@ -9,22 +9,5 @@
"devDependencies": {
"@types/node": "^20.11.17",
"tsx": "^4.7.1"
},
"overrides": {
"glob": ">=11.1.0",
"tar": ">=7.5.10",
"minimatch": ">=10.2.4",
"diff": ">=8.0.3",
"@isaacs/brace-expansion": ">=5.0.1",
"@babel/traverse": ">=7.23.2",
"ws": ">=7.5.10",
"http-proxy-middleware": ">=2.0.9",
"tar-fs": ">=2.1.4",
"webpack-dev-middleware": ">=5.3.4",
"braces": ">=3.0.3",
"axios": ">=0.30.2",
"webpack": ">=5.94.0",
"serve-static": ">=1.16.0",
"path-to-regexp": ">=0.1.12"
}
}
}

View file

@ -1,11 +0,0 @@
-- DropIndex
DROP INDEX IF EXISTS "LiteLLM_MCPServerTable_approval_status_idx";
-- AlterTable
ALTER TABLE "LiteLLM_MCPServerTable" DROP COLUMN IF EXISTS "approval_status",
DROP COLUMN IF EXISTS "review_notes",
DROP COLUMN IF EXISTS "reviewed_at",
DROP COLUMN IF EXISTS "source_url",
DROP COLUMN IF EXISTS "submitted_at",
DROP COLUMN IF EXISTS "submitted_by";

View file

@ -0,0 +1,13 @@
-- Restore fields dropped by 20260311180521_schema_sync on LiteLLM_MCPServerTable
-- AlterTable
ALTER TABLE "LiteLLM_MCPServerTable"
ADD COLUMN IF NOT EXISTS "source_url" TEXT,
ADD COLUMN IF NOT EXISTS "approval_status" TEXT DEFAULT 'active',
ADD COLUMN IF NOT EXISTS "submitted_by" TEXT,
ADD COLUMN IF NOT EXISTS "submitted_at" TIMESTAMP(3),
ADD COLUMN IF NOT EXISTS "reviewed_at" TIMESTAMP(3),
ADD COLUMN IF NOT EXISTS "review_notes" TEXT;
-- CreateIndex
CREATE INDEX IF NOT EXISTS "LiteLLM_MCPServerTable_approval_status_idx"
ON "LiteLLM_MCPServerTable"("approval_status");

View file

@ -320,11 +320,14 @@ model LiteLLM_MCPServerTable {
is_byok Boolean @default(false)
byok_description String[] @default([])
byok_api_key_help_url String?
approval_status String @default("approved")
source_url String?
approval_status String? @default("active")
submitted_by String?
submitted_at DateTime?
reviewed_at DateTime?
review_notes String?
@@index([approval_status])
}
// Per-user BYOK credentials for MCP servers

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm-proxy-extras"
version = "0.4.60"
version = "0.4.62"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
authors = ["BerriAI"]
readme = "README.md"
@ -22,7 +22,7 @@ requires = ["poetry-core"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "0.4.60"
version = "0.4.62"
version_files = [
"pyproject.toml:version",
"../requirements.txt:litellm-proxy-extras==",

View file

@ -32,6 +32,7 @@ from litellm.llms.base_llm.bridges.completion_transformation import (
)
from litellm.types.llms.openai import (
ChatCompletionAnnotation,
ChatCompletionReasoningItem,
ChatCompletionToolParamFunctionChunk,
Reasoning,
ResponsesAPIOptionalRequestParams,
@ -55,6 +56,61 @@ if TYPE_CHECKING:
)
def _get_reasoning_items(
msg: "AllMessageValues",
) -> List[ChatCompletionReasoningItem]:
"""Extract reasoning_items from a message dict with proper typing."""
items = msg.get("reasoning_items") # type: ignore[union-attr]
if items:
return items # type: ignore[return-value]
return []
def _build_reasoning_item(
item_id: str,
encrypted_content: Optional[str],
summary_raw: Any,
) -> Dict[str, Any]:
"""Build a ChatCompletionReasoningItem-shaped dict from raw response data.
Handles both pydantic objects (attribute access) and plain dicts.
"""
summary: List[Dict[str, Any]] = []
for s in summary_raw or []:
if isinstance(s, dict):
summary.append(
{"type": s.get("type", "summary_text"), "text": s.get("text", "")}
)
else:
summary.append(
{
"type": getattr(s, "type", "summary_text"),
"text": getattr(s, "text", ""),
}
)
return {
"id": item_id,
"type": "reasoning",
"encrypted_content": encrypted_content,
"summary": summary,
}
def _reasoning_item_to_response_input(
r_item: Union[ChatCompletionReasoningItem, Dict[str, Any]]
) -> Dict[str, Any]:
"""Convert a stored ChatCompletionReasoningItem back to a Responses API input item."""
r_input: Dict[str, Any] = {
"type": "reasoning",
"id": r_item.get("id") or f"rs_{id(r_item)}",
# summary is always required by the Responses API, even when empty
"summary": r_item.get("summary") or [],
}
if r_item.get("encrypted_content"):
r_input["encrypted_content"] = r_item["encrypted_content"]
return r_input
class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
"""
Handler for transforming /chat/completions api requests to litellm.responses requests
@ -202,10 +258,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
}
)
elif role == "assistant" and tool_calls and isinstance(tool_calls, list):
for r_item in _get_reasoning_items(msg):
input_items.append(_reasoning_item_to_response_input(r_item))
for tool_call in tool_calls:
function = tool_call.get("function")
if function:
input_tool_call = {
input_tool_call: Dict[str, Any] = {
"type": "function_call",
"call_id": tool_call["id"],
}
@ -217,7 +275,9 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
else:
raise ValueError(f"tool call not supported: {tool_call}")
elif content is not None:
# Regular user/assistant message
if role == "assistant":
for r_item in _get_reasoning_items(msg):
input_items.append(_reasoning_item_to_response_input(r_item))
input_items.append(
{
"type": "message",
@ -411,6 +471,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
choices: List[Choices] = []
index = 0
reasoning_content: Optional[str] = None
pending_reasoning_item: Optional[Dict[str, Any]] = None
# Collect all tool calls to put them in a single choice
# (Chat Completions API expects all tool calls in one message)
@ -419,9 +480,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
for item in output_items:
if isinstance(item, ResponseReasoningItem):
for summary_item in item.summary:
response_text = getattr(summary_item, "text", "")
reasoning_content = response_text if response_text else ""
pending_reasoning_item = _build_reasoning_item(
item_id=item.id,
encrypted_content=getattr(item, "encrypted_content", None),
summary_raw=item.summary,
)
reasoning_content = " ".join(
s["text"]
for s in pending_reasoning_item["summary"]
if s.get("text")
)
elif isinstance(item, ResponseOutputMessage):
for content in item.content:
@ -436,6 +504,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
content=response_text if response_text else "",
reasoning_content=reasoning_content,
annotations=annotations,
reasoning_items=cast(
Optional[List[ChatCompletionReasoningItem]],
[pending_reasoning_item]
if pending_reasoning_item is not None
else None,
),
)
choices.append(
@ -446,7 +520,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
)
)
reasoning_content = None # flush reasoning content
reasoning_content = None # flush
pending_reasoning_item = None # flush
index += 1
elif isinstance(item, ResponseFunctionToolCall):
@ -489,11 +564,18 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
content=None,
tool_calls=accumulated_tool_calls,
reasoning_content=reasoning_content,
reasoning_items=cast(
Optional[List[ChatCompletionReasoningItem]],
[pending_reasoning_item]
if pending_reasoning_item is not None
else None,
),
)
choices.append(
Choices(message=msg, finish_reason="tool_calls", index=index)
)
reasoning_content = None
pending_reasoning_item = None
return choices
@ -1232,6 +1314,25 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
finish_reason = "tool_calls" if has_function_calls else "stop"
# Extract reasoning items with encrypted_content for round-tripping
completed_reasoning_items: Optional[List[Dict[str, Any]]] = None
for item in output_items:
if not isinstance(item, dict) or item.get("type") != "reasoning":
continue
if completed_reasoning_items is None:
completed_reasoning_items = []
completed_reasoning_items.append(
_build_reasoning_item(
item_id=item.get("id", ""),
encrypted_content=item.get("encrypted_content"),
summary_raw=item.get("summary"),
)
)
completed_reasoning_items_typed = cast(
Optional[List[ChatCompletionReasoningItem]],
completed_reasoning_items,
)
usage = None
if response_data.get("usage"):
from litellm.responses.utils import ResponseAPILoggingUtils
@ -1245,7 +1346,10 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
choices=[
StreamingChoices(
index=0,
delta=Delta(content=""),
delta=Delta(
content="",
reasoning_items=completed_reasoning_items_typed,
),
finish_reason=finish_reason,
)
],

View file

@ -1402,6 +1402,9 @@ DEFAULT_SHARED_HEALTH_CHECK_TTL = int(
DEFAULT_SHARED_HEALTH_CHECK_LOCK_TTL = int(
os.getenv("DEFAULT_SHARED_HEALTH_CHECK_LOCK_TTL", 60)
) # 1 minute - TTL for health check lock
DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER = (
2 # health state is stale after interval * this
)
PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS = int(
os.getenv("PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS", 9)
)

View file

@ -65,6 +65,17 @@ def _get_cached_end_user_id_for_cost_tracking():
class PrometheusLogger(CustomLogger):
# Class variables or attributes
@staticmethod
def get_instance() -> Optional["PrometheusLogger"]:
"""Find the PrometheusLogger instance from litellm.callbacks, if registered."""
import litellm
for cb in litellm.callbacks:
if isinstance(cb, PrometheusLogger):
return cb
return None
def __init__( # noqa: PLR0915
self,
**kwargs,
@ -180,6 +191,31 @@ class PrometheusLogger(CustomLogger):
),
)
# Remaining Budget for Org
self.litellm_remaining_org_budget_metric = self._gauge_factory(
"litellm_remaining_org_budget_metric",
"Remaining budget for org",
labelnames=self.get_labels_for_metric(
"litellm_remaining_org_budget_metric"
),
)
# Max Budget for Org
self.litellm_org_max_budget_metric = self._gauge_factory(
"litellm_org_max_budget_metric",
"Maximum budget set for org",
labelnames=self.get_labels_for_metric("litellm_org_max_budget_metric"),
)
# Org Budget Reset At
self.litellm_org_budget_remaining_hours_metric = self._gauge_factory(
"litellm_org_budget_remaining_hours_metric",
"Remaining hours for org budget to be reset",
labelnames=self.get_labels_for_metric(
"litellm_org_budget_remaining_hours_metric"
),
)
# Remaining Budget for API Key
self.litellm_remaining_api_key_budget_metric = self._gauge_factory(
"litellm_remaining_api_key_budget_metric",
@ -440,6 +476,76 @@ class PrometheusLogger(CustomLogger):
labelnames=[],
)
########################################
# Managed Batch Metrics
########################################
self.litellm_managed_batch_created_total = self._counter_factory(
name="litellm_managed_batch_created_total",
documentation="Total number of managed batches created",
labelnames=[
"model",
"api_provider",
"user",
"user_email",
"api_key_alias",
],
)
self.litellm_managed_file_size_bytes = self._gauge_factory(
"litellm_managed_file_size_bytes",
"Size of the most recent managed batch file in bytes (last-seen value per label combination)",
labelnames=["purpose", "file_type", "model", "api_provider", "user"],
)
self.litellm_managed_batch_duration_seconds = self._histogram_factory(
"litellm_managed_batch_duration_seconds",
"Duration of completed managed batches in seconds (completed_at - created_at)",
labelnames=["model", "api_provider"],
buckets=BATCH_DURATION_BUCKETS,
)
self.litellm_managed_file_created_total = self._counter_factory(
name="litellm_managed_file_created_total",
documentation="Total number of managed files created",
labelnames=[
"model",
"api_provider",
"user",
"user_email",
"api_key_alias",
],
)
self.litellm_managed_file_deleted_total = self._counter_factory(
name="litellm_managed_file_deleted_total",
documentation="Total number of managed file deletions (success or blocked)",
labelnames=["result"],
)
self.litellm_check_batch_cost_jobs_polled = self._gauge_factory(
"litellm_check_batch_cost_jobs_polled",
"Number of unprocessed batches found by the last CheckBatchCost poll",
labelnames=[],
)
self.litellm_check_batch_cost_jobs_processed_total = self._counter_factory(
name="litellm_check_batch_cost_jobs_processed_total",
documentation="Total number of batches successfully cost-tracked by CheckBatchCost",
labelnames=["model", "api_provider"],
)
self.litellm_check_batch_cost_errors_total = self._counter_factory(
name="litellm_check_batch_cost_errors_total",
documentation="Total number of errors in CheckBatchCost by error type",
labelnames=["error_type"],
)
self.litellm_check_batch_cost_last_run_timestamp = self._gauge_factory(
"litellm_check_batch_cost_last_run_timestamp",
"Unix timestamp of the last CheckBatchCost job run",
labelnames=[],
)
except Exception as e:
print_verbose(f"Got exception on init prometheus client {str(e)}")
raise e
@ -922,6 +1028,9 @@ class PrometheusLogger(CustomLogger):
user_api_team_alias = standard_logging_payload["metadata"][
"user_api_key_team_alias"
]
user_api_key_org_id = standard_logging_payload["metadata"].get(
"user_api_key_org_id"
)
output_tokens = standard_logging_payload["completion_tokens"]
tokens_used = standard_logging_payload["total_tokens"]
response_cost = standard_logging_payload["response_cost"]
@ -1030,6 +1139,7 @@ class PrometheusLogger(CustomLogger):
litellm_params=litellm_params,
response_cost=response_cost,
user_id=user_id,
user_api_key_org_id=user_api_key_org_id,
)
# set proxy virtual key rpm/tpm metrics
@ -1185,6 +1295,7 @@ class PrometheusLogger(CustomLogger):
litellm_params: dict,
response_cost: float,
user_id: Optional[str] = None,
user_api_key_org_id: Optional[str] = None,
):
_metadata = litellm_params.get("metadata") or {}
_team_spend = _metadata.get("user_api_key_team_spend", None)
@ -1217,12 +1328,16 @@ class PrometheusLogger(CustomLogger):
user_max_budget=_user_max_budget,
response_cost=response_cost,
),
self._set_org_budget_metrics_after_api_request(
org_id=user_api_key_org_id,
response_cost=response_cost,
),
return_exceptions=True,
)
for i, r in enumerate(results):
if isinstance(r, Exception):
verbose_logger.debug(
f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user'][i]} failed: {r}"
f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user', 'org'][i]} failed: {r}"
)
def _increment_top_level_request_and_spend_metrics(
@ -1411,6 +1526,9 @@ class PrometheusLogger(CustomLogger):
user_api_team_alias = standard_logging_payload["metadata"][
"user_api_key_team_alias"
]
user_api_key_org_id = standard_logging_payload["metadata"].get(
"user_api_key_org_id"
)
try:
self.litellm_llm_api_failed_requests_metric.labels(
@ -1426,6 +1544,10 @@ class PrometheusLogger(CustomLogger):
),
).inc()
self.set_llm_deployment_failure_metrics(kwargs)
await self._set_org_budget_metrics_after_api_request(
org_id=user_api_key_org_id,
response_cost=0,
)
except Exception as e:
verbose_logger.exception(
"prometheus Layer Error(): Exception occured - {}".format(str(e))
@ -2158,6 +2280,127 @@ class PrometheusLogger(CustomLogger):
except Exception as e:
verbose_logger.debug(f"Error recording guardrail metrics: {str(e)}")
########################################
# Managed Batch Metric Recording Methods
########################################
def record_managed_batch_created(
self,
model: Optional[str],
api_provider: Optional[str],
user: Optional[str],
user_email: Optional[str],
api_key_alias: Optional[str],
):
try:
self.litellm_managed_batch_created_total.labels(
model=model,
api_provider=api_provider,
user=user,
user_email=user_email,
api_key_alias=api_key_alias,
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording batch created metric: {e}")
def record_managed_file_size(
self,
size_bytes: int,
purpose: str,
file_type: str,
model: Optional[str] = None,
api_provider: Optional[str] = None,
user: Optional[str] = None,
):
"""Record the size of a managed file. Uses a gauge (last-seen value per label combination)."""
try:
self.litellm_managed_file_size_bytes.labels(
purpose=purpose,
file_type=file_type,
model=model or "",
api_provider=api_provider or "",
user=user or "",
).set(size_bytes)
except Exception as e:
verbose_logger.warning(f"Error recording file size metric: {e}")
def record_managed_batch_duration(
self,
duration_seconds: float,
model: Optional[str] = None,
api_provider: Optional[str] = None,
):
try:
self.litellm_managed_batch_duration_seconds.labels(
model=model or "",
api_provider=api_provider or "",
).observe(duration_seconds)
except Exception as e:
verbose_logger.warning(f"Error recording batch duration metric: {e}")
def record_managed_file_created(
self,
model: Optional[str],
api_provider: Optional[str],
user: Optional[str],
user_email: Optional[str],
api_key_alias: Optional[str],
):
try:
self.litellm_managed_file_created_total.labels(
model=model,
api_provider=api_provider,
user=user,
user_email=user_email,
api_key_alias=api_key_alias,
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording file created metric: {e}")
def record_managed_file_deleted(self, result: str):
"""Record a managed file deletion attempt. result is 'success' or 'blocked'."""
try:
self.litellm_managed_file_deleted_total.labels(result=result).inc()
except Exception as e:
verbose_logger.warning(f"Error recording file deleted metric: {e}")
def record_check_batch_cost_run(
self,
jobs_polled: int,
processed_models: Optional[List[Tuple[Optional[str], Optional[str]]]] = None,
):
"""
Record CheckBatchCost polling metrics.
Args:
jobs_polled: Number of unprocessed batches found
processed_models: List of (model, api_provider) tuples for processed jobs
"""
import time
try:
self.litellm_check_batch_cost_last_run_timestamp.set(time.time())
self.litellm_check_batch_cost_jobs_polled.set(jobs_polled)
if processed_models:
for model, api_provider in processed_models:
self.litellm_check_batch_cost_jobs_processed_total.labels(
model=model or "",
api_provider=api_provider or "",
).inc()
except Exception as e:
verbose_logger.warning(f"Error recording check batch cost metrics: {e}")
def record_check_batch_cost_error(self, error_type: str):
try:
self.litellm_check_batch_cost_errors_total.labels(
error_type=error_type,
).inc()
except Exception as e:
verbose_logger.warning(
f"Error recording check batch cost error metric: {e}"
)
@staticmethod
def _get_exception_class_name(exception: Exception) -> str:
exception_class_name = ""
@ -2534,6 +2777,35 @@ class PrometheusLogger(CustomLogger):
data_type="users",
)
async def _initialize_org_budget_metrics(self):
"""
Initialize org budget metrics by reusing the generic pagination logic.
"""
from litellm.proxy.proxy_server import prisma_client
if prisma_client is None:
verbose_logger.debug(
"Prometheus: skipping org metrics initialization, DB not initialized"
)
return
async def fetch_orgs(page_size: int, page: int) -> Tuple[list, Optional[int]]:
skip = (page - 1) * page_size
orgs = await prisma_client.db.litellm_organizationtable.find_many(
skip=skip,
take=page_size,
order={"created_at": "desc"},
include={"litellm_budget_table": True},
)
total_count = await prisma_client.db.litellm_organizationtable.count()
return orgs, total_count
await self._initialize_budget_metrics(
data_fetch_function=fetch_orgs,
set_metrics_function=self._set_org_list_budget_metrics,
data_type="orgs",
)
async def initialize_remaining_budget_metrics(self):
"""
Handler for initializing remaining budget metrics for all teams to avoid metric discrepancies.
@ -2568,10 +2840,11 @@ class PrometheusLogger(CustomLogger):
"""
Helper to initialize remaining budget metrics for all teams, API keys, and users.
"""
verbose_logger.debug("Emitting key, team, user budget metrics....")
verbose_logger.debug("Emitting key, team, user, org budget metrics....")
await self._initialize_team_budget_metrics()
await self._initialize_api_key_budget_metrics()
await self._initialize_user_budget_metrics()
await self._initialize_org_budget_metrics()
await self._initialize_user_and_team_count_metrics()
async def _initialize_user_and_team_count_metrics(self):
@ -2627,6 +2900,20 @@ class PrometheusLogger(CustomLogger):
for user in users:
self._set_user_budget_metrics(user)
async def _set_org_list_budget_metrics(self, orgs: list):
"""Helper function to set budget metrics for a list of orgs"""
for org in orgs:
budget_table = getattr(org, "litellm_budget_table", None)
self._set_org_budget_metrics(
org_id=org.organization_id or "",
org_alias=org.organization_alias or "",
spend=org.spend or 0.0,
max_budget=budget_table.max_budget if budget_table else None,
budget_reset_at=getattr(budget_table, "budget_reset_at", None)
if budget_table
else None,
)
async def _set_team_budget_metrics_after_api_request(
self,
user_api_team: Optional[str],
@ -2748,6 +3035,113 @@ class PrometheusLogger(CustomLogger):
)
)
async def _set_org_budget_metrics_after_api_request(
self,
org_id: Optional[str],
response_cost: float,
):
"""
Set org budget metrics after an LLM API request
- Fetches org info via cache (get_org_object)
- Sets org budget metrics
"""
if not org_id:
return
from litellm.proxy.auth.auth_checks import get_org_object
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
if prisma_client is None:
return
try:
org_info = await get_org_object(
org_id=org_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
include_budget_table=True,
)
except Exception as e:
verbose_logger.debug(
f"[Non-Blocking] Prometheus: Error getting org info: {str(e)}"
)
return
if org_info is None:
return
org_alias = org_info.organization_alias or ""
_total_org_spend = (org_info.spend or 0.0) + response_cost
budget_table = org_info.litellm_budget_table
max_budget = budget_table.max_budget if budget_table else None
budget_reset_at = (
getattr(budget_table, "budget_reset_at", None) if budget_table else None
)
self._set_org_budget_metrics(
org_id=org_id,
org_alias=org_alias,
spend=_total_org_spend,
max_budget=max_budget,
budget_reset_at=budget_reset_at,
)
def _set_org_budget_metrics(
self,
org_id: str,
org_alias: str,
spend: float,
max_budget: Optional[float],
budget_reset_at: Optional[datetime],
):
"""
Set org budget metrics for a single org
- Remaining Budget
- Max Budget
- Budget Reset At
"""
enum_values = UserAPIKeyLabelValues(
org_id=org_id,
org_alias=org_alias,
)
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_remaining_org_budget_metric"
),
enum_values=enum_values,
)
self.litellm_remaining_org_budget_metric.labels(**_labels).set(
self._safe_get_remaining_budget(
max_budget=max_budget,
spend=spend,
)
)
if max_budget is not None:
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_org_max_budget_metric"
),
enum_values=enum_values,
)
self.litellm_org_max_budget_metric.labels(**_labels).set(max_budget)
if budget_reset_at is not None:
_labels = prometheus_label_factory(
supported_enum_labels=self.get_labels_for_metric(
metric_name="litellm_org_budget_remaining_hours_metric"
),
enum_values=enum_values,
)
self.litellm_org_budget_remaining_hours_metric.labels(**_labels).set(
self._get_remaining_hours_for_budget_reset(
budget_reset_at=budget_reset_at
)
)
def _set_key_budget_metrics(self, user_api_key_dict: UserAPIKeyAuth):
"""
Set virtual key budget metrics

View file

@ -158,17 +158,17 @@ def get_llm_provider( # noqa: PLR0915
): # handle scenario where model="azure/*" and custom_llm_provider="azure"
model = custom_llm_provider + "/" + model
# Native OpenRouter models have IDs like "openrouter/free" where the
# "openrouter/" prefix is part of the actual model name on the API.
# When called from a bridge (e.g. anthropic_messages adapter),
# custom_llm_provider is already resolved, so return early to prevent
# the provider-list stripping below from removing the prefix.
# OpenRouter: when the router/proxy already set custom_llm_provider,
# the model may still carry LiteLLM's "openrouter/" routing prefix.
# Native IDs like "openrouter/auto" must stay intact for the API; IDs
# like "openrouter/anthropic/claude-3.5-sonnet" must become
# "anthropic/claude-3.5-sonnet" (OpenRouter expects provider/model).
if custom_llm_provider == "openrouter" and model.startswith("openrouter/"):
remainder = model[len("openrouter/") :]
if "/" in remainder:
return remainder, custom_llm_provider, dynamic_api_key, api_base
return model, custom_llm_provider, dynamic_api_key, api_base
if api_key and api_key.startswith("os.environ/"):
dynamic_api_key = get_secret_str(api_key)
# Check JSON-configured providers FIRST (before enum-based provider_list)
provider_prefix = model.split("/", 1)[0]
if len(model.split("/")) > 1 and JSONProviderRegistry.exists(provider_prefix):

View file

@ -3000,9 +3000,10 @@ class Logging(LiteLLMLoggingBaseClass):
litellm_call_id=self.model_call_details["litellm_call_id"],
print_verbose=print_verbose,
)
if (
callable(callback) and customLogger is not None
): # custom logger functions
if callable(callback): # custom logger functions
global customLogger
if customLogger is None:
customLogger = CustomLogger()
customLogger.log_event(
kwargs=self.model_call_details,
response_obj=result,
@ -3143,9 +3144,10 @@ class Logging(LiteLLMLoggingBaseClass):
start_time=start_time,
end_time=end_time,
) # type: ignore
if (
callable(callback) and customLogger is not None
): # custom logger functions
if callable(callback): # custom logger functions
global customLogger
if customLogger is None:
customLogger = CustomLogger()
await customLogger.async_log_event(
kwargs=self.model_call_details,
response_obj=result,

View file

@ -831,6 +831,10 @@ class CustomStreamWrapper:
"annotations" in model_response.choices[0].delta
and model_response.choices[0].delta.annotations is not None
)
or (
getattr(model_response.choices[0].delta, "reasoning_items", None)
is not None
)
):
return True
else:
@ -2174,12 +2178,16 @@ class CustomStreamWrapper:
None,
)
if _deferred_cb is not None:
# Proxy has post-call guardrails — let the closure
# run guardrails on the assembled response, then
# fire logging with guardrail_information populated.
self.logging_obj._on_deferred_stream_complete = None # type: ignore[attr-defined]
asyncio.create_task(
_deferred_cb(complete_streaming_response, cache_hit)
# Proxy has post-call guardrails. Store the assembled
# response so the outer streaming consumer
# (ProxyLogging.async_post_call_streaming_iterator_hook)
# can fire the deferred callback AFTER all guardrail
# end-of-stream blocks complete. Scheduling here via
# create_task would race with unified_guardrail's
# end-of-stream block for short-stream providers.
self.logging_obj._deferred_stream_complete_args = ( # type: ignore[attr-defined]
complete_streaming_response,
cache_hit,
)
else:
asyncio.create_task(

View file

@ -234,37 +234,13 @@ class A2AGuardrailHandler(BaseTranslation):
then the combined guardrailed text is written into the first chunk that had text
and all other text parts in other chunks are cleared (in-place).
"""
from litellm.llms.a2a.common_utils import extract_text_from_a2a_response
# Parse each item; keep alignment with responses_so_far (None where unparseable)
parsed: List[Optional[Dict[str, Any]]] = [None] * len(responses_so_far)
for i, item in enumerate(responses_so_far):
if isinstance(item, dict):
obj = item
elif isinstance(item, str):
try:
obj = json.loads(item.strip())
except (json.JSONDecodeError, TypeError):
continue
else:
continue
if isinstance(obj.get("result"), dict):
parsed[i] = obj
valid_parsed = [(i, obj) for i, obj in enumerate(parsed) if obj is not None]
parsed, valid_parsed = self._parse_streaming_responses(responses_so_far)
if not valid_parsed:
return responses_so_far
# Collect text from each chunk in order (by original index in responses_so_far)
text_parts: List[str] = []
chunk_indices_with_text: List[int] = [] # indices into valid_parsed
for idx, (orig_i, obj) in enumerate(valid_parsed):
t = extract_text_from_a2a_response(obj)
if t:
text_parts.append(t)
chunk_indices_with_text.append(orig_i)
combined_text = "".join(text_parts)
combined_text, chunk_indices_with_text = self._collect_text_from_parsed_chunks(
valid_parsed
)
if not combined_text:
return responses_so_far
@ -337,6 +313,43 @@ class A2AGuardrailHandler(BaseTranslation):
return responses_so_far
def _parse_streaming_responses(
self,
responses_so_far: List[Any],
) -> Tuple[List[Optional[Dict[str, Any]]], List[Tuple[int, Dict[str, Any]]]]:
"""Parse JSON-RPC items, returning aligned parsed list and valid entries."""
parsed: List[Optional[Dict[str, Any]]] = [None] * len(responses_so_far)
for i, item in enumerate(responses_so_far):
if isinstance(item, dict):
obj = item
elif isinstance(item, str):
try:
obj = json.loads(item.strip())
except (json.JSONDecodeError, TypeError):
continue
else:
continue
if isinstance(obj.get("result"), dict):
parsed[i] = obj
valid_parsed = [(i, obj) for i, obj in enumerate(parsed) if obj is not None]
return parsed, valid_parsed
def _collect_text_from_parsed_chunks(
self,
valid_parsed: List[Tuple[int, Dict[str, Any]]],
) -> Tuple[str, List[int]]:
"""Collect text from parsed chunks, returning combined text and indices."""
from litellm.llms.a2a.common_utils import extract_text_from_a2a_response
text_parts: List[str] = []
chunk_indices_with_text: List[int] = []
for _idx, (orig_i, obj) in enumerate(valid_parsed):
t = extract_text_from_a2a_response(obj)
if t:
text_parts.append(t)
chunk_indices_with_text.append(orig_i)
return "".join(text_parts), chunk_indices_with_text
def _extract_texts_from_result(
self,
result: Dict[str, Any],

View file

@ -277,82 +277,35 @@ class AnthropicMessagesHandler(BaseTranslation):
images_to_check: List[str] = []
tool_calls_to_check: List[ChatCompletionToolCallChunk] = []
task_mappings: List[Tuple[int, Optional[int]]] = []
# Track (content_index, None) for each text
# Handle both dict and object responses
response_content: List[Any] = []
if isinstance(response, dict):
response_content = response.get("content", []) or []
elif hasattr(response, "content"):
content = getattr(response, "content", None)
response_content = content or []
else:
response_content = []
response_content = self._get_response_content(response)
if not response_content:
return response
# Step 1: Extract all text content and tool calls from response
for content_idx, content_block in enumerate(response_content):
# Handle both dict and Pydantic object content blocks
block_dict: Dict[str, Any] = {}
if isinstance(content_block, dict):
block_type = content_block.get("type")
block_dict = cast(Dict[str, Any], content_block)
elif hasattr(content_block, "type"):
block_type = getattr(content_block, "type", None)
# Convert Pydantic object to dict for processing
if hasattr(content_block, "model_dump"):
block_dict = content_block.model_dump()
else:
block_dict = {
"type": block_type,
"text": getattr(content_block, "text", None),
}
else:
continue
if block_type in ["text", "tool_use"]:
self._extract_output_text_and_images(
content_block=block_dict,
content_idx=content_idx,
texts_to_check=texts_to_check,
images_to_check=images_to_check,
task_mappings=task_mappings,
tool_calls_to_check=tool_calls_to_check,
)
self._extract_from_content_blocks(
response_content,
texts_to_check,
images_to_check,
task_mappings,
tool_calls_to_check,
)
# Step 2: Apply guardrail to all texts in batch
if texts_to_check or tool_calls_to_check:
# Use the real request_data if provided (proxy path), otherwise
# create a standalone dict (SDK / direct-call path).
if request_data is None:
request_data = {"response": response}
else:
if "response" not in request_data:
request_data["response"] = response
request_data = self._prepare_request_data(
request_data,
response,
user_api_key_dict,
key="response",
)
# Add user API key metadata with prefixed keys
if "litellm_metadata" not in request_data:
user_metadata = self.transform_user_api_key_dict_to_metadata(
user_api_key_dict
)
if user_metadata:
request_data["litellm_metadata"] = user_metadata
inputs = GenericGuardrailAPIInputs(texts=texts_to_check)
if images_to_check:
inputs["images"] = images_to_check
if tool_calls_to_check:
inputs["tool_calls"] = tool_calls_to_check
# Include model information from the response if available
response_model = None
if isinstance(response, dict):
response_model = response.get("model")
elif hasattr(response, "model"):
response_model = getattr(response, "model", None)
if response_model:
inputs["model"] = response_model
inputs = self._build_guardrail_inputs(
texts_to_check,
images_to_check,
tool_calls_to_check,
response,
)
guardrailed_inputs = await guardrail_to_apply.apply_guardrail(
inputs=inputs,
@ -440,6 +393,95 @@ class AnthropicMessagesHandler(BaseTranslation):
)
return responses_so_far
def _prepare_request_data(
self,
request_data: Optional[dict],
response: Any,
user_api_key_dict: Optional[Any],
key: str,
) -> dict:
"""Ensure request_data has the response/responses_so_far key and metadata."""
if request_data is None:
request_data = {key: response}
else:
if key not in request_data:
request_data[key] = response
if "litellm_metadata" not in request_data:
user_metadata = self.transform_user_api_key_dict_to_metadata(
user_api_key_dict
)
if user_metadata:
request_data["litellm_metadata"] = user_metadata
return request_data
@staticmethod
def _get_response_content(response: Any) -> List[Any]:
"""Extract content list from a dict or object response."""
if isinstance(response, dict):
return response.get("content", []) or []
elif hasattr(response, "content"):
return getattr(response, "content", None) or []
return []
def _extract_from_content_blocks(
self,
response_content: List[Any],
texts_to_check: List[str],
images_to_check: List[str],
task_mappings: List[Tuple[int, Optional[int]]],
tool_calls_to_check: List["ChatCompletionToolCallChunk"],
) -> None:
"""Extract text, images, and tool calls from content blocks."""
for content_idx, content_block in enumerate(response_content):
block_dict: Dict[str, Any] = {}
if isinstance(content_block, dict):
block_type = content_block.get("type")
block_dict = cast(Dict[str, Any], content_block)
elif hasattr(content_block, "type"):
block_type = getattr(content_block, "type", None)
if hasattr(content_block, "model_dump"):
block_dict = content_block.model_dump()
else:
block_dict = {
"type": block_type,
"text": getattr(content_block, "text", None),
}
else:
continue
if block_type in ["text", "tool_use"]:
self._extract_output_text_and_images(
content_block=block_dict,
content_idx=content_idx,
texts_to_check=texts_to_check,
images_to_check=images_to_check,
task_mappings=task_mappings,
tool_calls_to_check=tool_calls_to_check,
)
@staticmethod
def _build_guardrail_inputs(
texts_to_check: List[str],
images_to_check: List[str],
tool_calls_to_check: List["ChatCompletionToolCallChunk"],
response: Any,
) -> "GenericGuardrailAPIInputs":
"""Build GenericGuardrailAPIInputs with optional images, tool calls, model."""
inputs = GenericGuardrailAPIInputs(texts=texts_to_check)
if images_to_check:
inputs["images"] = images_to_check
if tool_calls_to_check:
inputs["tool_calls"] = tool_calls_to_check
response_model = None
if isinstance(response, dict):
response_model = response.get("model")
elif hasattr(response, "model"):
response_model = getattr(response, "model", None)
if response_model:
inputs["model"] = response_model
return inputs
def get_streaming_string_so_far(self, responses_so_far: List[Any]) -> str:
"""
Parse streaming responses and extract accumulated text content.

View file

@ -1421,6 +1421,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
):
optional_params["metadata"] = {"user_id": _litellm_metadata["user_id"]}
## Ensure metadata only contains user_id (only documented field in Anthropic Messages API)
if "metadata" in optional_params and isinstance(
optional_params["metadata"], dict
):
_user_id = optional_params["metadata"].get("user_id")
if _user_id is not None:
optional_params["metadata"] = {"user_id": _user_id}
else:
optional_params.pop("metadata")
# Remove internal LiteLLM parameters that should not be sent to Anthropic API
optional_params.pop("is_vertex_request", None)

View file

@ -1,10 +1,15 @@
from typing import Optional, Union
from typing import Any, Coroutine, Dict, Optional, Union, cast
import httpx
from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI
from litellm._logging import verbose_logger
from litellm.llms.azure.common_utils import BaseAzureLLM
from litellm.llms.openai.fine_tuning.handler import OpenAIFineTuningAPI
from litellm.llms.openai.fine_tuning.handler import (
OpenAIFineTuningAPI,
_litellm_fine_tuning_job_from_response,
)
from litellm.types.utils import LiteLLMFineTuningJob
class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
@ -12,6 +17,194 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
AzureOpenAI methods to support fine tuning, inherits from OpenAIFineTuningAPI.
"""
@staticmethod
def _ensure_training_type(create_fine_tuning_job_data: Dict[str, Any]) -> None:
"""
Azure requires trainingType in extra_body. Default to 1 (supervised) if omitted.
"""
extra_body = create_fine_tuning_job_data.get("extra_body") or {}
if not isinstance(extra_body, dict):
extra_body = {}
if extra_body.get("trainingType") is None:
extra_body["trainingType"] = 1
create_fine_tuning_job_data["extra_body"] = extra_body
verbose_logger.debug(
"Azure fine-tuning: defaulting trainingType=1 (supervised)"
)
async def acreate_fine_tuning_job(
self,
create_fine_tuning_job_data: dict,
openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI],
) -> LiteLLMFineTuningJob:
response = await openai_client.fine_tuning.jobs.create(
**create_fine_tuning_job_data
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
async def acancel_fine_tuning_job(
self,
fine_tuning_job_id: str,
openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI],
) -> LiteLLMFineTuningJob:
response = await openai_client.fine_tuning.jobs.cancel(
fine_tuning_job_id=fine_tuning_job_id
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
async def aretrieve_fine_tuning_job(
self,
fine_tuning_job_id: str,
openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI],
) -> LiteLLMFineTuningJob:
response = await openai_client.fine_tuning.jobs.retrieve(
fine_tuning_job_id=fine_tuning_job_id
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
def create_fine_tuning_job(
self,
_is_async: bool,
create_fine_tuning_job_data: dict,
api_key: Optional[str],
api_base: Optional[str],
api_version: Optional[str],
timeout: Union[float, httpx.Timeout],
max_retries: Optional[int],
organization: Optional[str],
client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = None,
) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]:
self._ensure_training_type(create_fine_tuning_job_data)
openai_client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = self.get_openai_client(
api_key=api_key,
api_base=api_base,
timeout=timeout,
max_retries=max_retries,
organization=organization,
client=client,
_is_async=_is_async,
api_version=api_version,
)
if openai_client is None:
raise ValueError(
"Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment."
)
if _is_async is True:
if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)):
raise ValueError(
"OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client."
)
return self.acreate_fine_tuning_job(
create_fine_tuning_job_data=create_fine_tuning_job_data,
openai_client=openai_client,
)
verbose_logger.debug(
"creating fine tuning job, args= %s", create_fine_tuning_job_data
)
response = cast(OpenAI, openai_client).fine_tuning.jobs.create(
**create_fine_tuning_job_data
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
def cancel_fine_tuning_job(
self,
_is_async: bool,
fine_tuning_job_id: str,
api_key: Optional[str],
api_base: Optional[str],
api_version: Optional[str],
timeout: Union[float, httpx.Timeout],
max_retries: Optional[int],
organization: Optional[str],
client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = None,
) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]:
openai_client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = self.get_openai_client(
api_key=api_key,
api_base=api_base,
timeout=timeout,
max_retries=max_retries,
organization=organization,
client=client,
_is_async=_is_async,
api_version=api_version,
)
if openai_client is None:
raise ValueError(
"Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment."
)
if _is_async is True:
if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)):
raise ValueError(
"OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client."
)
return self.acancel_fine_tuning_job(
fine_tuning_job_id=fine_tuning_job_id,
openai_client=openai_client,
)
response = cast(OpenAI, openai_client).fine_tuning.jobs.cancel(
fine_tuning_job_id=fine_tuning_job_id
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
def retrieve_fine_tuning_job(
self,
_is_async: bool,
fine_tuning_job_id: str,
api_key: Optional[str],
api_base: Optional[str],
api_version: Optional[str],
timeout: Union[float, httpx.Timeout],
max_retries: Optional[int],
organization: Optional[str],
client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = None,
) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]:
openai_client: Optional[
Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI]
] = self.get_openai_client(
api_key=api_key,
api_base=api_base,
timeout=timeout,
max_retries=max_retries,
organization=organization,
client=client,
_is_async=_is_async,
api_version=api_version,
)
if openai_client is None:
raise ValueError(
"Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment."
)
if _is_async is True:
if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)):
raise ValueError(
"OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client."
)
return self.aretrieve_fine_tuning_job(
fine_tuning_job_id=fine_tuning_job_id,
openai_client=openai_client,
)
response = cast(OpenAI, openai_client).fine_tuning.jobs.retrieve(
fine_tuning_job_id=fine_tuning_job_id
)
return _litellm_fine_tuning_job_from_response(response, is_azure=True)
def get_openai_client(
self,
api_key: Optional[str],

View file

@ -91,34 +91,6 @@ UNSUPPORTED_BEDROCK_CONVERSE_BETA_PATTERNS = [
"compact-2026-01-12", # The compact beta feature is not currently supported on the Converse and ConverseStream APIs
]
# Models that support Bedrock's native structured outputs API (outputConfig.textFormat)
# Uses substring matching against the Bedrock model ID
# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/structured-output.html
BEDROCK_NATIVE_STRUCTURED_OUTPUT_MODELS = {
# Anthropic Claude 4.5+
"claude-haiku-4-5",
"claude-sonnet-4-5",
"claude-opus-4-5",
"claude-opus-4-6",
# Qwen3
"qwen3",
# DeepSeek
"deepseek-v3.1",
# Gemma 3
"gemma-3",
# MiniMax
"minimax-m2",
# Mistral (magistral-small excluded: broken constrained decoding on Bedrock)
"ministral",
"mistral-large-3",
"voxtral",
# Moonshot
"kimi-k2",
# NVIDIA
"nemotron-nano",
# OpenAI (gpt-oss excluded: broken constrained decoding, works via tool-call fallback)
}
class AmazonConverseConfig(BaseConfig):
"""
@ -493,8 +465,7 @@ class AmazonConverseConfig(BaseConfig):
budget = thinking.get("budget_tokens")
if isinstance(budget, int) and budget < BEDROCK_MIN_THINKING_BUDGET_TOKENS:
verbose_logger.debug(
"Bedrock requires thinking.budget_tokens >= %d, got %d. "
"Clamping to minimum.",
"Bedrock requires thinking.budget_tokens >= %d, got %d. Clamping to minimum.",
BEDROCK_MIN_THINKING_BUDGET_TOKENS,
budget,
)
@ -763,10 +734,20 @@ class AmazonConverseConfig(BaseConfig):
return _tool
@staticmethod
def _supports_native_structured_outputs(model: str) -> bool:
"""Check if the Bedrock model supports native structured outputs (outputConfig.textFormat)."""
return any(
substring in model for substring in BEDROCK_NATIVE_STRUCTURED_OUTPUT_MODELS
def _supports_native_structured_outputs(
model: str, custom_llm_provider: Optional[str] = None
) -> bool:
"""Check if the Bedrock model supports native structured outputs (outputConfig.textFormat).
Delegates to the standard ``supports_native_structured_output`` utility
which looks up the flag in ``litellm.model_cost`` via
``_get_model_info_helper``.
Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/structured-output.html
"""
from litellm.utils import supports_native_structured_output
return supports_native_structured_output(
model=model, custom_llm_provider=custom_llm_provider
)
@staticmethod
@ -913,7 +894,9 @@ class AmazonConverseConfig(BaseConfig):
)
if param == "tool_choice":
_tool_choice_value = self.map_tool_choice_values(
model=model, tool_choice=value, drop_params=drop_params # type: ignore
model=model,
tool_choice=value,
drop_params=drop_params, # type: ignore
)
if _tool_choice_value is not None:
optional_params["tool_choice"] = _tool_choice_value
@ -1006,7 +989,10 @@ class AmazonConverseConfig(BaseConfig):
if "type" in value and value["type"] == "text":
return optional_params
if self._supports_native_structured_outputs(model) and json_schema is not None:
if (
self._supports_native_structured_outputs(model, self.custom_llm_provider)
and json_schema is not None
):
# Use Bedrock's native structured outputs API (outputConfig.textFormat)
# No synthetic tool injection, no fake_stream needed.
# Requires an explicit schema — json_object with no schema falls through

View file

@ -5,6 +5,7 @@ For vertex ai, check out the vertex_ai/files/handler.py file.
"""
import time
from typing import Any, List, Literal, Optional
from urllib.parse import urlparse
import httpx
from openai.types.file_deleted import FileDeleted
@ -209,27 +210,58 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig):
"""
Get the URL to retrieve a file from Google AI Studio.
We expect file_id to be the URI (e.g. https://generativelanguage.googleapis.com/v1beta/files/...)
as returned by the upload response.
Endpoint:
GET https://generativelanguage.googleapis.com/v1beta/{name=files/*}
The URL should look like:
https://generativelanguage.googleapis.com/v1beta/files/{file_id}?key=API_KEY
We expect file_id to be just the file identifier (e.g., files/abc123 or abc123)
as returned by the upload response. (If it's a full URL, extract the file name.)
"""
api_key = litellm_params.get("api_key") or self.get_api_key()
if not api_key:
raise ValueError("api_key is required")
if file_id.startswith("http"):
url = "{}?key={}".format(file_id, api_key)
else:
# Fallback for just file name (files/...)
api_base = (
self.get_api_base(litellm_params.get("api_base"))
or "https://generativelanguage.googleapis.com"
)
api_base = api_base.rstrip("/")
url = "{}/v1beta/{}?key={}".format(api_base, file_id, api_key)
file_part = self._normalize_gemini_file_id(file_id)
api_base = (
self.get_api_base(litellm_params.get("api_base"))
or "https://generativelanguage.googleapis.com"
)
api_base = api_base.rstrip("/")
url = f"{api_base}/v1beta/{file_part}?key={api_key}"
# Return empty params dict - API key is already in URL, no query params needed
return url, {}
def _normalize_gemini_file_id(self, file_id: str) -> str:
"""
Normalize file identifier into `files/{id}` form.
Supports:
- `abc123`
- `files/abc123`
- `https://generativelanguage.googleapis.com/v1beta/files/abc123`
"""
if file_id.startswith(("http://", "https://")):
parsed = urlparse(file_id)
path = parsed.path.lstrip("/")
files_index = path.find("files/")
if files_index != -1:
normalized_file_id = path[files_index:]
else:
normalized_file_id = path
else:
normalized_file_id = file_id
normalized_file_id = normalized_file_id.strip("/")
if not normalized_file_id.startswith("files/"):
normalized_file_id = f"files/{normalized_file_id}"
return normalized_file_id
def transform_retrieve_file_response(
self,
raw_response: httpx.Response,
@ -240,8 +272,9 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig):
Transform Gemini's file retrieval response into OpenAI-style FileObject
"""
try:
verbose_logger.debug(f"Retrieve file response: {raw_response.text}")
response_json = raw_response.json()
verbose_logger.debug(f"Response JSON: {response_json}")
# Map Gemini state to OpenAI status
gemini_state = response_json.get("state", "STATE_UNSPECIFIED")
# Explicitly type status as the Literal union

View file

@ -1,4 +1,4 @@
from typing import Any, Coroutine, Optional, Union, cast
from typing import Any, Coroutine, Dict, Optional, Union, cast
import httpx
from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI
@ -6,6 +6,55 @@ from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI
from litellm._logging import verbose_logger
from litellm.types.utils import LiteLLMFineTuningJob
_AZURE_STATUS_MAP = {
"pending": "queued",
"notRunning": "queued",
"running": "running",
"succeeded": "succeeded",
"failed": "failed",
"canceled": "cancelled",
"canceling": "cancelled",
}
# Note: Azure's "canceling" (in-progress) is mapped to "cancelled" (terminal)
# because LiteLLMFineTuningJob schema has no intermediate cancellation state.
def _normalize_fine_tuning_job_dict(
data: Dict[str, Any], is_azure: bool = False
) -> Dict[str, Any]:
"""
Normalize Azure OpenAI FineTuningJob response to match OpenAI schema.
Azure differences:
- organization_id: null → ""
- result_files: null → []
- status: mapped via _AZURE_STATUS_MAP
"""
if not is_azure:
return data
normalized = data.copy()
if normalized.get("organization_id") is None:
normalized["organization_id"] = ""
if normalized.get("result_files") is None:
normalized["result_files"] = []
status = normalized.get("status")
if status in _AZURE_STATUS_MAP:
normalized["status"] = _AZURE_STATUS_MAP[status]
return normalized
def _litellm_fine_tuning_job_from_response(
response: Any, is_azure: bool = False
) -> LiteLLMFineTuningJob:
return LiteLLMFineTuningJob(
**_normalize_fine_tuning_job_dict(response.model_dump(), is_azure=is_azure)
)
class OpenAIFineTuningAPI:
"""
@ -60,7 +109,7 @@ class OpenAIFineTuningAPI:
**create_fine_tuning_job_data
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)
def create_fine_tuning_job(
self,
@ -108,7 +157,7 @@ class OpenAIFineTuningAPI:
response = cast(OpenAI, openai_client).fine_tuning.jobs.create(
**create_fine_tuning_job_data
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)
async def acancel_fine_tuning_job(
self,
@ -118,7 +167,7 @@ class OpenAIFineTuningAPI:
response = await openai_client.fine_tuning.jobs.cancel(
fine_tuning_job_id=fine_tuning_job_id
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)
def cancel_fine_tuning_job(
self,
@ -164,7 +213,7 @@ class OpenAIFineTuningAPI:
response = cast(OpenAI, openai_client).fine_tuning.jobs.cancel(
fine_tuning_job_id=fine_tuning_job_id
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)
async def alist_fine_tuning_jobs(
self,
@ -229,7 +278,7 @@ class OpenAIFineTuningAPI:
response = await openai_client.fine_tuning.jobs.retrieve(
fine_tuning_job_id=fine_tuning_job_id
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)
def retrieve_fine_tuning_job(
self,
@ -275,4 +324,4 @@ class OpenAIFineTuningAPI:
response = cast(OpenAI, openai_client).fine_tuning.jobs.retrieve(
fine_tuning_job_id=fine_tuning_job_id
)
return LiteLLMFineTuningJob(**response.model_dump())
return _litellm_fine_tuning_job_from_response(response)

View file

@ -722,7 +722,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"anthropic.claude-haiku-4-5@20251001": {
"cache_creation_input_token_cost": 1.25e-06,
@ -745,7 +746,8 @@
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346,
"supports_native_streaming": true
"supports_native_streaming": true,
"supports_native_structured_output": true
},
"anthropic.claude-3-5-sonnet-20240620-v1:0": {
"input_cost_per_token": 3e-06,
@ -967,22 +969,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_native_structured_output": true
},
"anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -997,22 +996,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"global.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1027,22 +1023,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"us.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1057,22 +1050,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"eu.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1087,22 +1077,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"au.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1117,22 +1104,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-07,
"input_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1147,22 +1131,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"global.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-07,
"input_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1177,22 +1158,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"us.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
"cache_read_input_token_cost": 3.3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
"input_cost_per_token": 3.3e-06,
"input_cost_per_token_above_200k_tokens": 6.6e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.65e-05,
"output_cost_per_token_above_200k_tokens": 2.475e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1207,22 +1185,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"eu.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
"cache_read_input_token_cost": 3.3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
"input_cost_per_token": 3.3e-06,
"input_cost_per_token_above_200k_tokens": 6.6e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.65e-05,
"output_cost_per_token_above_200k_tokens": 2.475e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1237,22 +1212,19 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"au.anthropic.claude-sonnet-4-6": {
"cache_creation_input_token_cost": 4.125e-06,
"cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
"cache_read_input_token_cost": 3.3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
"input_cost_per_token": 3.3e-06,
"input_cost_per_token_above_200k_tokens": 6.6e-06,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.65e-05,
"output_cost_per_token_above_200k_tokens": 2.475e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -1267,7 +1239,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -1327,7 +1300,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_native_structured_output": true
},
"anthropic.claude-v1": {
"input_cost_per_token": 8e-06,
@ -1577,7 +1551,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"apac.anthropic.claude-3-sonnet-20240229-v1:0": {
"input_cost_per_token": 3e-06,
@ -1665,7 +1640,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"azure/ada": {
"input_cost_per_token": 1e-07,
@ -1831,7 +1807,7 @@
"cache_read_input_token_cost": 3e-07,
"input_cost_per_token": 3e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
@ -8503,18 +8479,14 @@
},
"claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-07,
"input_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "anthropic",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -8695,19 +8667,15 @@
},
"claude-opus-4-6": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "anthropic",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -8730,19 +8698,15 @@
},
"claude-opus-4-6-20260205": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 1e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "anthropic",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -11742,7 +11706,8 @@
"output_cost_per_token": 1.68e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"supports_native_structured_output": true
},
"deepseek.v3.2": {
"input_cost_per_token": 6.2e-07,
@ -12187,7 +12152,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"eu.anthropic.claude-3-5-sonnet-20240620-v1:0": {
"input_cost_per_token": 3e-06,
@ -12401,7 +12367,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"eu.meta.llama3-2-1b-instruct-v1:0": {
"input_cost_per_token": 1.3e-07,
@ -14631,18 +14598,6 @@
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"uses_embed_content": true
},
"vertex_ai/gemini-embedding-2-preview": {
"input_cost_per_token": 1.5e-07,
"litellm_provider": "vertex_ai",
"max_input_tokens": 8192,
"max_tokens": 8192,
"mode": "embedding",
"output_cost_per_token": 0,
"output_vector_size": 3072,
"source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal",
"supports_multimodal": true,
"uses_embed_content": true
},
"gemini/gemini-embedding-001": {
"input_cost_per_token": 1.5e-07,
"litellm_provider": "gemini",
@ -15931,6 +15886,55 @@
"supports_tool_choice": true,
"supports_vision": true
},
"gemini/lyria-3-clip-preview": {
"input_cost_per_token": 0,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_image": 0.04,
"output_cost_per_token": 0,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"audio"
],
"supports_audio_input": false,
"supports_audio_output": true,
"supports_function_calling": false,
"supports_prompt_caching": false,
"supports_response_schema": false,
"supports_system_messages": false,
"supports_vision": false,
"supports_web_search": false
},
"gemini/lyria-3-pro-preview": {
"input_cost_per_token": 0,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 0,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"supported_modalities": [
"text"
],
"supported_output_modalities": [
"audio"
],
"supports_audio_input": false,
"supports_audio_output": true,
"supports_function_calling": false,
"supports_prompt_caching": false,
"supports_response_schema": false,
"supports_system_messages": false,
"supports_vision": false,
"supports_web_search": false
},
"gemini/veo-2.0-generate-001": {
"litellm_provider": "gemini",
"max_input_tokens": 1024,
@ -16770,7 +16774,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"global.anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -16822,7 +16827,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"global.amazon.nova-2-lite-v1:0": {
"cache_read_input_token_cost": 7.5e-08,
@ -20551,7 +20557,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"jp.anthropic.claude-haiku-4-5-20251001-v1:0": {
"cache_creation_input_token_cost": 1.375e-06,
@ -20573,7 +20580,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"lambda_ai/deepseek-llama3.3-70b": {
"input_cost_per_token": 2e-07,
@ -21268,7 +21276,8 @@
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"minimax.minimax-m2.1": {
"input_cost_per_token": 3e-07,
@ -21424,7 +21433,8 @@
"mode": "chat",
"output_cost_per_token": 2e-07,
"supports_function_calling": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral.ministral-3-3b-instruct": {
"input_cost_per_token": 1e-07,
@ -21435,7 +21445,8 @@
"mode": "chat",
"output_cost_per_token": 1e-07,
"supports_function_calling": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral.ministral-3-8b-instruct": {
"input_cost_per_token": 1.5e-07,
@ -21446,7 +21457,8 @@
"mode": "chat",
"output_cost_per_token": 1.5e-07,
"supports_function_calling": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral.mistral-7b-instruct-v0:2": {
"input_cost_per_token": 1.5e-07,
@ -21488,7 +21500,8 @@
"mode": "chat",
"output_cost_per_token": 1.5e-06,
"supports_function_calling": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral.mistral-small-2402-v1:0": {
"input_cost_per_token": 1e-06,
@ -21519,7 +21532,8 @@
"mode": "chat",
"output_cost_per_token": 4e-08,
"supports_audio_input": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral.voxtral-small-24b-2507": {
"input_cost_per_token": 1e-07,
@ -21530,7 +21544,8 @@
"mode": "chat",
"output_cost_per_token": 3e-07,
"supports_audio_input": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"mistral/codestral-2405": {
"input_cost_per_token": 1e-06,
@ -22217,7 +22232,8 @@
"mode": "chat",
"output_cost_per_token": 2.5e-06,
"supports_reasoning": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"moonshotai.kimi-k2.5": {
"input_cost_per_token": 6e-07,
@ -23092,7 +23108,8 @@
"supports_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"source": "https://aws.amazon.com/bedrock/pricing/"
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_native_structured_output": true
},
"o1": {
"cache_read_input_token_cost": 7.5e-06,
@ -26176,7 +26193,8 @@
"output_cost_per_token": 1.8e-06,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"supports_native_structured_output": true
},
"qwen.qwen3-235b-a22b-2507-v1:0": {
"input_cost_per_token": 2.2e-07,
@ -26188,7 +26206,8 @@
"output_cost_per_token": 8.8e-07,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"supports_native_structured_output": true
},
"qwen.qwen3-coder-30b-a3b-v1:0": {
"input_cost_per_token": 1.5e-07,
@ -26200,7 +26219,8 @@
"output_cost_per_token": 6e-07,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"supports_native_structured_output": true
},
"qwen.qwen3-32b-v1:0": {
"input_cost_per_token": 1.5e-07,
@ -26212,7 +26232,8 @@
"output_cost_per_token": 6e-07,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_tool_choice": true
"supports_tool_choice": true,
"supports_native_structured_output": true
},
"qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.5e-07,
@ -26223,7 +26244,8 @@
"mode": "chat",
"output_cost_per_token": 1.2e-06,
"supports_function_calling": true,
"supports_system_messages": true
"supports_system_messages": true,
"supports_native_structured_output": true
},
"qwen.qwen3-vl-235b-a22b": {
"input_cost_per_token": 5.3e-07,
@ -26235,7 +26257,8 @@
"output_cost_per_token": 2.66e-06,
"supports_function_calling": true,
"supports_system_messages": true,
"supports_vision": true
"supports_vision": true,
"supports_native_structured_output": true
},
"qwen.qwen3-coder-next": {
"input_cost_per_token": 5e-07,
@ -28260,7 +28283,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"us.anthropic.claude-3-5-sonnet-20240620-v1:0": {
"input_cost_per_token": 3e-06,
@ -28418,7 +28442,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"au.anthropic.claude-haiku-4-5-20251001-v1:0": {
"cache_creation_input_token_cost": 1.375e-06,
@ -28439,7 +28464,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
"tool_use_system_prompt_tokens": 346,
"supports_native_structured_output": true
},
"us.anthropic.claude-opus-4-20250514-v1:0": {
"cache_creation_input_token_cost": 1.875e-05,
@ -28491,7 +28517,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_native_structured_output": true
},
"global.anthropic.claude-opus-4-5-20251101-v1:0": {
"cache_creation_input_token_cost": 6.25e-06,
@ -28517,7 +28544,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_native_structured_output": true
},
"eu.anthropic.claude-opus-4-5-20251101-v1:0": {
"cache_creation_input_token_cost": 6.25e-06,
@ -28543,7 +28571,8 @@
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 159
"tool_use_system_prompt_tokens": 159,
"supports_native_structured_output": true
},
"us.anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
@ -30323,18 +30352,14 @@
},
"vertex_ai/claude-opus-4-6": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "vertex_ai-anthropic_models",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -30353,18 +30378,14 @@
},
"vertex_ai/claude-opus-4-6@default": {
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.25e-05,
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1e-06,
"input_cost_per_token": 5e-06,
"input_cost_per_token_above_200k_tokens": 1e-05,
"litellm_provider": "vertex_ai-anthropic_models",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"output_cost_per_token_above_200k_tokens": 3.75e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@ -30409,18 +30430,14 @@
},
"vertex_ai/claude-sonnet-4-6": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-07,
"input_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "vertex_ai-anthropic_models",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@ -36830,6 +36847,38 @@
"supports_audio_input": true,
"supports_audio_output": true
},
"gemini-3.1-flash-live-preview": {
"input_cost_per_audio_token": 3e-06,
"input_cost_per_image_token": 1e-06,
"input_cost_per_token": 7.5e-07,
"input_cost_per_video_per_second": 3.3333333333333335e-05,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"output_cost_per_audio_token": 1.2e-05,
"output_cost_per_token": 4.5e-06,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio",
"video"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_vision": true,
"supports_web_search": true
},
"gemini/gemini-2.5-flash-native-audio-latest": {
"input_cost_per_audio_token": 1e-06,
"input_cost_per_token": 3e-07,
@ -36908,6 +36957,40 @@
"tpm": 250000,
"rpm": 10
},
"gemini/gemini-3.1-flash-live-preview": {
"input_cost_per_audio_token": 3e-06,
"input_cost_per_image_token": 1e-06,
"input_cost_per_token": 7.5e-07,
"input_cost_per_video_per_second": 3.3333333333333335e-05,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"output_cost_per_audio_token": 1.2e-05,
"output_cost_per_token": 4.5e-06,
"source": "https://ai.google.dev/gemini-api/docs/pricing",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio",
"video"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_vision": true,
"supports_web_search": true,
"tpm": 250000,
"rpm": 10
},
"gemini-2.5-flash-preview-tts": {
"input_cost_per_token": 3e-07,
"litellm_provider": "gemini",
@ -37153,18 +37236,14 @@
},
"vertex_ai/claude-sonnet-4-6@default": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_200k_tokens": 7.5e-06,
"cache_read_input_token_cost": 3e-07,
"cache_read_input_token_cost_above_200k_tokens": 6e-07,
"input_cost_per_token": 3e-06,
"input_cost_per_token_above_200k_tokens": 6e-06,
"litellm_provider": "vertex_ai-anthropic_models",
"max_input_tokens": 200000,
"max_input_tokens": 1000000,
"max_output_tokens": 64000,
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,

Some files were not shown because too many files have changed in this diff Show more