mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
fix: prompt registry
This commit is contained in:
parent
26cd194d97
commit
56fab12fbe
3344 changed files with 397984 additions and 58287 deletions
File diff suppressed because it is too large
Load diff
|
|
@ -8,12 +8,13 @@ redis==5.2.1
|
||||||
redisvl==0.4.1
|
redisvl==0.4.1
|
||||||
anthropic
|
anthropic
|
||||||
orjson==3.10.12 # fast /embedding responses
|
orjson==3.10.12 # fast /embedding responses
|
||||||
pydantic==2.10.2
|
pydantic==2.11.0
|
||||||
google-cloud-aiplatform==1.43.0
|
google-cloud-aiplatform==1.43.0
|
||||||
google-cloud-iam==2.19.1
|
google-cloud-iam==2.19.1
|
||||||
fastapi-sso==0.16.0
|
fastapi-sso==0.16.0
|
||||||
uvloop==0.21.0
|
uvloop==0.21.0
|
||||||
mcp==1.10.1 # for MCP server
|
mcp==1.25.0 # for MCP server
|
||||||
semantic_router==0.1.10 # for auto-routing with litellm
|
semantic_router==0.1.10 # for auto-routing with litellm
|
||||||
fastuuid==0.12.0
|
fastuuid==0.12.0
|
||||||
responses==0.25.7 # for proxy client tests
|
responses==0.25.7 # for proxy client tests
|
||||||
|
pytest-retry==1.6.3 # for automatic test retries
|
||||||
|
|
@ -48,7 +48,7 @@ dist/
|
||||||
build/
|
build/
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
.DS_Store
|
.DS_Store
|
||||||
node_modules/
|
**/node_modules
|
||||||
*.log
|
*.log
|
||||||
.env
|
.env
|
||||||
.env.local
|
.env.local
|
||||||
|
|
|
||||||
111
.gitguardian.yaml
Normal file
111
.gitguardian.yaml
Normal file
|
|
@ -0,0 +1,111 @@
|
||||||
|
version: 2
|
||||||
|
|
||||||
|
secret:
|
||||||
|
# Exclude files and paths by globbing
|
||||||
|
ignored_paths:
|
||||||
|
- "**/*.whl"
|
||||||
|
- "**/*.pyc"
|
||||||
|
- "**/__pycache__/**"
|
||||||
|
- "**/node_modules/**"
|
||||||
|
- "**/dist/**"
|
||||||
|
- "**/build/**"
|
||||||
|
- "**/.git/**"
|
||||||
|
- "**/venv/**"
|
||||||
|
- "**/.venv/**"
|
||||||
|
|
||||||
|
# Large data/metadata files that don't need scanning
|
||||||
|
- "**/model_prices_and_context_window*.json"
|
||||||
|
- "**/*_metadata/*.txt"
|
||||||
|
- "**/tokenizers/*.json"
|
||||||
|
- "**/tokenizers/*"
|
||||||
|
- "miniconda.sh"
|
||||||
|
|
||||||
|
# Build outputs and static assets
|
||||||
|
- "litellm/proxy/_experimental/out/**"
|
||||||
|
- "ui/litellm-dashboard/public/**"
|
||||||
|
- "**/swagger/*.js"
|
||||||
|
- "**/*.woff"
|
||||||
|
- "**/*.woff2"
|
||||||
|
- "**/*.avif"
|
||||||
|
- "**/*.webp"
|
||||||
|
|
||||||
|
# Test data files
|
||||||
|
- "**/tests/**/data_map.txt"
|
||||||
|
- "tests/**/*.txt"
|
||||||
|
|
||||||
|
# Documentation and other non-code files
|
||||||
|
- "docs/**"
|
||||||
|
- "**/*.md"
|
||||||
|
- "**/*.lock"
|
||||||
|
- "poetry.lock"
|
||||||
|
- "package-lock.json"
|
||||||
|
|
||||||
|
# Ignore security incidents with the SHA256 of the occurrence (false positives)
|
||||||
|
ignored_matches:
|
||||||
|
# === Current detected false positives (SHA-based) ===
|
||||||
|
|
||||||
|
# gcs_pub_sub_body - folder name, not a password
|
||||||
|
- name: GCS pub/sub test folder name
|
||||||
|
match: 75f377c456eede69e5f6e47399ccee6016a2a93cc5dd11db09cc5b1359ae569a
|
||||||
|
|
||||||
|
# os.environ/APORIA_API_KEY_1 - environment variable reference
|
||||||
|
- name: Environment variable reference APORIA_API_KEY_1
|
||||||
|
match: e2ddeb8b88eca97a402559a2be2117764e11c074d86159ef9ad2375dea188094
|
||||||
|
|
||||||
|
# os.environ/APORIA_API_KEY_2 - environment variable reference
|
||||||
|
- name: Environment variable reference APORIA_API_KEY_2
|
||||||
|
match: 09aa39a29e050b86603aa55138af1ff08fb86a4582aa965c1bd0672e1575e052
|
||||||
|
|
||||||
|
# oidc/circleci_v2/ - test authentication path, not a secret
|
||||||
|
- name: OIDC CircleCI test path
|
||||||
|
match: feb3475e1f89a65b7b7815ac4ec597e18a9ec1847742ad445c36ca617b536e15
|
||||||
|
|
||||||
|
# text-davinci-003 - OpenAI model identifier, not a secret
|
||||||
|
- name: OpenAI model identifier text-davinci-003
|
||||||
|
match: c489000cf6c7600cee0eefb80ad0965f82921cfb47ece880930eb7e7635cf1f1
|
||||||
|
|
||||||
|
# Base64 Basic Auth in test_pass_through_endpoints.py - test fixture, not a real secret
|
||||||
|
- name: Test Base64 Basic Auth header in pass_through_endpoints test
|
||||||
|
match: 61bac0491f395040617df7ef6d06029eac4d92a4457ac784978db80d97be1ae0
|
||||||
|
|
||||||
|
# PostgreSQL password "postgres" in CI configs - standard test database password
|
||||||
|
- name: Test PostgreSQL password in CI configurations
|
||||||
|
match: 6e0d657eb1f0fbc40cf0b8f3c3873ef627cc9cb7c4108d1c07d979c04bc8a4bb
|
||||||
|
|
||||||
|
# Bearer token in locustfile.py - test/example API key for load testing
|
||||||
|
- name: Test Bearer token in locustfile load test
|
||||||
|
match: 2a0abc2b0c3c1760a51ffcdf8d6b1d384cef69af740504b1cfa82dd70cdc7ff9
|
||||||
|
|
||||||
|
# Inkeep API key in docusaurus.config.js - public documentation site key
|
||||||
|
- name: Inkeep API key in documentation config
|
||||||
|
match: c366657791bfb5fc69045ec11d49452f09a0aebbc8648f94e2469b4025e29a75
|
||||||
|
|
||||||
|
# Langfuse credentials in test_completion.py - test credentials for integration test
|
||||||
|
- name: Langfuse test credentials in test_completion
|
||||||
|
match: c39310f68cc3d3e22f7b298bb6353c4f45759adcc37080d8b7f4e535d3cfd7f4
|
||||||
|
|
||||||
|
# Test password "sk-1234" in e2e test fixtures - test fixture, not a real secret
|
||||||
|
- name: Test password in e2e test fixtures
|
||||||
|
match: ce32b547202e209ec1dd50107b64be4cfcf2eb15c3b4f8e9dc611ef747af634f
|
||||||
|
|
||||||
|
# === Preventive patterns for test keys (pattern-based) ===
|
||||||
|
|
||||||
|
# Test API keys (124 instances across 45 files)
|
||||||
|
- name: Test API keys with sk-test prefix
|
||||||
|
match: sk-test-
|
||||||
|
|
||||||
|
# Mock API keys
|
||||||
|
- name: Mock API keys with sk-mock prefix
|
||||||
|
match: sk-mock-
|
||||||
|
|
||||||
|
# Fake API keys
|
||||||
|
- name: Fake API keys with sk-fake prefix
|
||||||
|
match: sk-fake-
|
||||||
|
|
||||||
|
# Generic test API key patterns
|
||||||
|
- name: Test API key patterns
|
||||||
|
match: test-api-key
|
||||||
|
|
||||||
|
- name: Short fake sk keys (1–9 digits only)
|
||||||
|
match: \bsk-\d{1,9}\b
|
||||||
|
|
||||||
38
.github/ISSUE_TEMPLATE/bug_report.yml
vendored
38
.github/ISSUE_TEMPLATE/bug_report.yml
vendored
|
|
@ -7,6 +7,16 @@ body:
|
||||||
attributes:
|
attributes:
|
||||||
value: |
|
value: |
|
||||||
Thanks for taking the time to fill out this bug report!
|
Thanks for taking the time to fill out this bug report!
|
||||||
|
|
||||||
|
**💡 Tip:** See our [Troubleshooting Guide](https://docs.litellm.ai/docs/troubleshoot) for what information to include.
|
||||||
|
- type: checkboxes
|
||||||
|
id: duplicate-check
|
||||||
|
attributes:
|
||||||
|
label: Check for existing issues
|
||||||
|
description: Please search to see if an issue already exists for the bug you encountered.
|
||||||
|
options:
|
||||||
|
- label: I have searched the existing issues and checked that my issue is not a duplicate.
|
||||||
|
required: true
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: what-happened
|
id: what-happened
|
||||||
attributes:
|
attributes:
|
||||||
|
|
@ -16,6 +26,21 @@ body:
|
||||||
value: "A bug happened!"
|
value: "A bug happened!"
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
|
- type: textarea
|
||||||
|
id: steps-to-reproduce
|
||||||
|
attributes:
|
||||||
|
label: Steps to Reproduce
|
||||||
|
description: Please provide detailed steps to reproduce this bug(A curl/python code to reproduce the bug)
|
||||||
|
placeholder: |
|
||||||
|
1. config.yaml file/ .env file/ etc.
|
||||||
|
2. Run the following code...
|
||||||
|
3. Observe the error...
|
||||||
|
value: |
|
||||||
|
1.
|
||||||
|
2.
|
||||||
|
3.
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: logs
|
id: logs
|
||||||
attributes:
|
attributes:
|
||||||
|
|
@ -23,13 +48,16 @@ body:
|
||||||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||||
render: shell
|
render: shell
|
||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: ml-ops-team
|
id: component
|
||||||
attributes:
|
attributes:
|
||||||
label: Are you a ML Ops Team?
|
label: What part of LiteLLM is this about?
|
||||||
description: This helps us prioritize your requests correctly
|
|
||||||
options:
|
options:
|
||||||
- "No"
|
- ''
|
||||||
- "Yes"
|
- "SDK (litellm Python package)"
|
||||||
|
- "Proxy"
|
||||||
|
- "UI Dashboard"
|
||||||
|
- "Docs"
|
||||||
|
- "Other"
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
- type: input
|
- type: input
|
||||||
|
|
|
||||||
21
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
21
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
|
|
@ -7,6 +7,14 @@ body:
|
||||||
attributes:
|
attributes:
|
||||||
value: |
|
value: |
|
||||||
Thanks for making LiteLLM better!
|
Thanks for making LiteLLM better!
|
||||||
|
- type: checkboxes
|
||||||
|
id: duplicate-check
|
||||||
|
attributes:
|
||||||
|
label: Check for existing issues
|
||||||
|
description: Please search to see if an issue already exists for the feature you are requesting.
|
||||||
|
options:
|
||||||
|
- label: I have searched the existing issues and checked that my issue is not a duplicate.
|
||||||
|
required: true
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: the-feature
|
id: the-feature
|
||||||
attributes:
|
attributes:
|
||||||
|
|
@ -22,6 +30,19 @@ body:
|
||||||
description: Please outline the motivation for the proposal. Is your feature request related to a specific problem? e.g., "I'm working on X and would like Y to be possible". If this is related to another GitHub issue, please link here too.
|
description: Please outline the motivation for the proposal. Is your feature request related to a specific problem? e.g., "I'm working on X and would like Y to be possible". If this is related to another GitHub issue, please link here too.
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
|
- type: dropdown
|
||||||
|
id: component
|
||||||
|
attributes:
|
||||||
|
label: What part of LiteLLM is this about?
|
||||||
|
options:
|
||||||
|
- ''
|
||||||
|
- "SDK (litellm Python package)"
|
||||||
|
- "Proxy"
|
||||||
|
- "UI Dashboard"
|
||||||
|
- "Docs"
|
||||||
|
- "Other"
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: hiring-interest
|
id: hiring-interest
|
||||||
attributes:
|
attributes:
|
||||||
|
|
|
||||||
|
|
@ -40,38 +40,33 @@ outputs:
|
||||||
runs:
|
runs:
|
||||||
using: composite
|
using: composite
|
||||||
steps:
|
steps:
|
||||||
|
- name: Helm | Setup
|
||||||
|
uses: azure/setup-helm@v4
|
||||||
|
with:
|
||||||
|
version: v3.20.0
|
||||||
|
|
||||||
- name: Helm | Login
|
- name: Helm | Login
|
||||||
shell: bash
|
shell: bash
|
||||||
run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }}
|
run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }}
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
- name: Helm | Dependency
|
- name: Helm | Dependency
|
||||||
if: inputs.update_dependencies == 'true'
|
if: inputs.update_dependencies == 'true'
|
||||||
shell: bash
|
shell: bash
|
||||||
run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
|
run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }}
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
- name: Helm | Package
|
- name: Helm | Package
|
||||||
shell: bash
|
shell: bash
|
||||||
run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }}
|
run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }}
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
- name: Helm | Push
|
- name: Helm | Push
|
||||||
shell: bash
|
shell: bash
|
||||||
run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }}
|
run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }}
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
- name: Helm | Logout
|
- name: Helm | Logout
|
||||||
shell: bash
|
shell: bash
|
||||||
run: helm registry logout ${{ inputs.registry }}
|
run: helm registry logout ${{ inputs.registry }}
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
- name: Helm | Output
|
- name: Helm | Output
|
||||||
id: output
|
id: output
|
||||||
shell: bash
|
shell: bash
|
||||||
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
|
run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT
|
||||||
|
|
|
||||||
24
.github/pull_request_template.md
vendored
24
.github/pull_request_template.md
vendored
|
|
@ -1,7 +1,3 @@
|
||||||
## Title
|
|
||||||
|
|
||||||
<!-- e.g. "Implement user authentication feature" -->
|
|
||||||
|
|
||||||
## Relevant issues
|
## Relevant issues
|
||||||
|
|
||||||
<!-- e.g. "Fixes #000" -->
|
<!-- e.g. "Fixes #000" -->
|
||||||
|
|
@ -11,10 +7,26 @@
|
||||||
**Please complete all items before asking a LiteLLM maintainer to review your PR**
|
**Please complete all items before asking a LiteLLM maintainer to review your PR**
|
||||||
|
|
||||||
- [ ] I have Added testing in the [`tests/litellm/`](https://github.com/BerriAI/litellm/tree/main/tests/litellm) directory, **Adding at least 1 test is a hard requirement** - [see details](https://docs.litellm.ai/docs/extras/contributing_code)
|
- [ ] I have Added testing in the [`tests/litellm/`](https://github.com/BerriAI/litellm/tree/main/tests/litellm) directory, **Adding at least 1 test is a hard requirement** - [see details](https://docs.litellm.ai/docs/extras/contributing_code)
|
||||||
- [ ] I have added a screenshot of my new test passing locally
|
|
||||||
- [ ] My PR passes all unit tests on [`make test-unit`](https://docs.litellm.ai/docs/extras/contributing_code)
|
- [ ] My PR passes all unit tests on [`make test-unit`](https://docs.litellm.ai/docs/extras/contributing_code)
|
||||||
- [ ] My PR's scope is as isolated as possible, it only solves 1 specific problem
|
- [ ] My PR's scope is as isolated as possible, it only solves 1 specific problem
|
||||||
|
- [ ] I have requested a Greptile review by commenting `@greptileai` and received a **Confidence Score of at least 4/5** before requesting a maintainer review
|
||||||
|
|
||||||
|
## CI (LiteLLM team)
|
||||||
|
|
||||||
|
> **CI status guideline:**
|
||||||
|
>
|
||||||
|
> - 50-55 passing tests: main is stable with minor issues.
|
||||||
|
> - 45-49 passing tests: acceptable but needs attention
|
||||||
|
> - <= 40 passing tests: unstable; be careful with your merges and assess the risk.
|
||||||
|
|
||||||
|
- [ ] **Branch creation CI run**
|
||||||
|
Link:
|
||||||
|
|
||||||
|
- [ ] **CI run for the last commit**
|
||||||
|
Link:
|
||||||
|
|
||||||
|
- [ ] **Merge / cherry-pick CI run**
|
||||||
|
Links:
|
||||||
|
|
||||||
## Type
|
## Type
|
||||||
|
|
||||||
|
|
@ -29,5 +41,3 @@
|
||||||
✅ Test
|
✅ Test
|
||||||
|
|
||||||
## Changes
|
## Changes
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
29
.github/workflows/check_duplicate_issues.yml
vendored
Normal file
29
.github/workflows/check_duplicate_issues.yml
vendored
Normal file
|
|
@ -0,0 +1,29 @@
|
||||||
|
name: Check Duplicate Issues
|
||||||
|
|
||||||
|
on:
|
||||||
|
issues:
|
||||||
|
types: [opened, edited]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
check-duplicate:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
issues: write
|
||||||
|
contents: read
|
||||||
|
steps:
|
||||||
|
- name: Check for potential duplicates
|
||||||
|
uses: wow-actions/potential-duplicates@v1
|
||||||
|
with:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
label: potential-duplicate
|
||||||
|
threshold: 0.6
|
||||||
|
reaction: eyes
|
||||||
|
comment: |
|
||||||
|
**⚠️ Potential duplicate detected**
|
||||||
|
|
||||||
|
This issue appears similar to existing issue(s):
|
||||||
|
{{#issues}}
|
||||||
|
- [#{{number}}]({{html_url}}) - {{title}} ({{accuracy}}% similar)
|
||||||
|
{{/issues}}
|
||||||
|
|
||||||
|
Please review the linked issue(s) to see if they address your concern. If this is not a duplicate, please provide additional context to help us understand the difference.
|
||||||
43
.github/workflows/create_daily_staging_branch.yml
vendored
Normal file
43
.github/workflows/create_daily_staging_branch.yml
vendored
Normal file
|
|
@ -0,0 +1,43 @@
|
||||||
|
name: Create Daily Staging Branch
|
||||||
|
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: '0 0,12 * * *' # Runs every 12 hours at midnight and noon UTC
|
||||||
|
workflow_dispatch: # Allow manual trigger
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
create-staging-branch:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v3
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- name: Create daily staging branch
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
run: |
|
||||||
|
# Configure Git user
|
||||||
|
git config user.name "github-actions[bot]"
|
||||||
|
git config user.email "github-actions[bot]@users.noreply.github.com"
|
||||||
|
|
||||||
|
# Generate branch name with MM_DD_YYYY format
|
||||||
|
BRANCH_NAME="litellm_oss_staging_$(date +'%m_%d_%Y')"
|
||||||
|
echo "Creating branch: $BRANCH_NAME"
|
||||||
|
|
||||||
|
# Fetch all branches
|
||||||
|
git fetch --all
|
||||||
|
|
||||||
|
# Check if the branch already exists
|
||||||
|
if git show-ref --verify --quiet refs/remotes/origin/$BRANCH_NAME; then
|
||||||
|
echo "Branch $BRANCH_NAME already exists. Skipping creation."
|
||||||
|
else
|
||||||
|
echo "Creating new branch: $BRANCH_NAME"
|
||||||
|
# Create the new branch from main
|
||||||
|
git checkout -b $BRANCH_NAME origin/main
|
||||||
|
# Push the new branch
|
||||||
|
git push origin $BRANCH_NAME
|
||||||
|
echo "Successfully created and pushed branch: $BRANCH_NAME"
|
||||||
|
fi
|
||||||
50
.github/workflows/ghcr_deploy.yml
vendored
50
.github/workflows/ghcr_deploy.yml
vendored
|
|
@ -5,6 +5,7 @@ on:
|
||||||
inputs:
|
inputs:
|
||||||
tag:
|
tag:
|
||||||
description: "The tag version you want to build"
|
description: "The tag version you want to build"
|
||||||
|
required: true
|
||||||
release_type:
|
release_type:
|
||||||
description: "The release type you want to build. Can be 'latest', 'stable', 'dev', 'rc'"
|
description: "The release type you want to build. Can be 'latest', 'stable', 'dev', 'rc'"
|
||||||
type: string
|
type: string
|
||||||
|
|
@ -319,44 +320,37 @@ jobs:
|
||||||
run: |
|
run: |
|
||||||
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
|
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
|
||||||
|
|
||||||
- name: Get LiteLLM Latest Tag
|
# Sync Helm chart version with LiteLLM release version (1-1 versioning)
|
||||||
id: current_app_tag
|
# This allows users to easily map Helm chart versions to LiteLLM versions
|
||||||
|
# See: https://codefresh.io/docs/docs/ci-cd-guides/helm-best-practices/
|
||||||
|
- name: Calculate chart and app versions
|
||||||
|
id: chart_version
|
||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
LATEST_TAG=$(git describe --tags --exclude "*dev*" --abbrev=0)
|
INPUT_TAG="${{ github.event.inputs.tag }}"
|
||||||
if [ -z "${LATEST_TAG}" ]; then
|
RELEASE_TYPE="${{ github.event.inputs.release_type }}"
|
||||||
echo "latest_tag=latest" | tee -a $GITHUB_OUTPUT
|
|
||||||
else
|
# Chart version = LiteLLM version without 'v' prefix (Helm semver convention)
|
||||||
echo "latest_tag=${LATEST_TAG}" | tee -a $GITHUB_OUTPUT
|
# v1.81.0 -> 1.81.0, v1.81.0.rc.1 -> 1.81.0.rc.1
|
||||||
|
CHART_VERSION="${INPUT_TAG#v}"
|
||||||
|
|
||||||
|
# Add suffix for 'latest' releases (rc already has suffix in tag)
|
||||||
|
if [ "$RELEASE_TYPE" = "latest" ]; then
|
||||||
|
CHART_VERSION="${CHART_VERSION}-latest"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
- name: Get last published chart version
|
# App version = Docker tag (keeps 'v' prefix to match Docker image tags)
|
||||||
id: current_version
|
APP_VERSION="${INPUT_TAG}"
|
||||||
shell: bash
|
|
||||||
run: |
|
|
||||||
CHART_LIST=$(helm show chart oci://${{ env.REGISTRY }}/${{ env.REPO_OWNER }}/${{ env.CHART_NAME }} 2>/dev/null || true)
|
|
||||||
if [ -z "${CHART_LIST}" ]; then
|
|
||||||
echo "current-version=0.1.0" | tee -a $GITHUB_OUTPUT
|
|
||||||
else
|
|
||||||
printf '%s' "${CHART_LIST}" | grep '^version:' | awk 'BEGIN{FS=":"}{print "current-version="$2}' | tr -d " " | tee -a $GITHUB_OUTPUT
|
|
||||||
fi
|
|
||||||
env:
|
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
# Automatically update the helm chart version one "patch" level
|
echo "version=${CHART_VERSION}" | tee -a $GITHUB_OUTPUT
|
||||||
- name: Bump release version
|
echo "app_version=${APP_VERSION}" | tee -a $GITHUB_OUTPUT
|
||||||
id: bump_version
|
|
||||||
uses: christian-draeger/increment-semantic-version@1.1.0
|
|
||||||
with:
|
|
||||||
current-version: ${{ steps.current_version.outputs.current-version || '0.1.0' }}
|
|
||||||
version-fragment: 'bug'
|
|
||||||
|
|
||||||
- uses: ./.github/actions/helm-oci-chart-releaser
|
- uses: ./.github/actions/helm-oci-chart-releaser
|
||||||
with:
|
with:
|
||||||
name: ${{ env.CHART_NAME }}
|
name: ${{ env.CHART_NAME }}
|
||||||
repository: ${{ env.REPO_OWNER }}
|
repository: ${{ env.REPO_OWNER }}
|
||||||
tag: ${{ github.event.inputs.chartVersion || steps.bump_version.outputs.next-version || '0.1.0' }}
|
tag: ${{ steps.chart_version.outputs.version }}
|
||||||
app_version: ${{ steps.current_app_tag.outputs.latest_tag }}
|
app_version: ${{ steps.chart_version.outputs.app_version }}
|
||||||
path: deploy/charts/${{ env.CHART_NAME }}
|
path: deploy/charts/${{ env.CHART_NAME }}
|
||||||
registry: ${{ env.REGISTRY }}
|
registry: ${{ env.REGISTRY }}
|
||||||
registry_username: ${{ github.actor }}
|
registry_username: ${{ github.actor }}
|
||||||
|
|
|
||||||
42
.github/workflows/ghcr_helm_deploy.yml
vendored
42
.github/workflows/ghcr_helm_deploy.yml
vendored
|
|
@ -1,10 +1,12 @@
|
||||||
# this workflow is triggered by an API call when there is a new PyPI release of LiteLLM
|
# Standalone workflow to publish LiteLLM Helm Chart
|
||||||
|
# Note: The main ghcr_deploy.yml workflow also publishes the Helm chart as part of a full release
|
||||||
name: Build, Publish LiteLLM Helm Chart. New Release
|
name: Build, Publish LiteLLM Helm Chart. New Release
|
||||||
on:
|
on:
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
chartVersion:
|
tag:
|
||||||
description: "Update the helm chart's version to this"
|
description: "LiteLLM version tag (e.g., v1.81.0)"
|
||||||
|
required: true
|
||||||
|
|
||||||
# Defines two custom environment variables for the workflow. Used for the Container registry domain, and a name for the Docker image that this workflow builds.
|
# Defines two custom environment variables for the workflow. Used for the Container registry domain, and a name for the Docker image that this workflow builds.
|
||||||
env:
|
env:
|
||||||
|
|
@ -31,24 +33,22 @@ jobs:
|
||||||
run: |
|
run: |
|
||||||
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
|
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
|
||||||
|
|
||||||
- name: Get LiteLLM Latest Tag
|
# Sync Helm chart version with LiteLLM release version (1-1 versioning)
|
||||||
id: current_app_tag
|
- name: Calculate chart and app versions
|
||||||
uses: WyriHaximus/github-action-get-previous-tag@v1.3.0
|
id: chart_version
|
||||||
|
|
||||||
- name: Get last published chart version
|
|
||||||
id: current_version
|
|
||||||
shell: bash
|
shell: bash
|
||||||
run: helm show chart oci://${{ env.REGISTRY }}/${{ env.REPO_OWNER }}/litellm-helm | grep '^version:' | awk 'BEGIN{FS=":"}{print "current-version="$2}' | tr -d " " | tee -a $GITHUB_OUTPUT
|
run: |
|
||||||
env:
|
INPUT_TAG="${{ github.event.inputs.tag }}"
|
||||||
HELM_EXPERIMENTAL_OCI: '1'
|
|
||||||
|
|
||||||
# Automatically update the helm chart version one "patch" level
|
# Chart version = LiteLLM version without 'v' prefix
|
||||||
- name: Bump release version
|
# v1.81.0 -> 1.81.0
|
||||||
id: bump_version
|
CHART_VERSION="${INPUT_TAG#v}"
|
||||||
uses: christian-draeger/increment-semantic-version@1.1.0
|
|
||||||
with:
|
# App version = Docker tag (keeps 'v' prefix)
|
||||||
current-version: ${{ steps.current_version.outputs.current-version || '0.1.0' }}
|
APP_VERSION="${INPUT_TAG}"
|
||||||
version-fragment: 'bug'
|
|
||||||
|
echo "version=${CHART_VERSION}" | tee -a $GITHUB_OUTPUT
|
||||||
|
echo "app_version=${APP_VERSION}" | tee -a $GITHUB_OUTPUT
|
||||||
|
|
||||||
- name: Lint helm chart
|
- name: Lint helm chart
|
||||||
run: helm lint deploy/charts/litellm-helm
|
run: helm lint deploy/charts/litellm-helm
|
||||||
|
|
@ -57,8 +57,8 @@ jobs:
|
||||||
with:
|
with:
|
||||||
name: litellm-helm
|
name: litellm-helm
|
||||||
repository: ${{ env.REPO_OWNER }}
|
repository: ${{ env.REPO_OWNER }}
|
||||||
tag: ${{ github.event.inputs.chartVersion || steps.bump_version.outputs.next-version || '0.1.0' }}
|
tag: ${{ steps.chart_version.outputs.version }}
|
||||||
app_version: ${{ steps.current_app_tag.outputs.tag || 'latest' }}
|
app_version: ${{ steps.chart_version.outputs.app_version }}
|
||||||
path: deploy/charts/litellm-helm
|
path: deploy/charts/litellm-helm
|
||||||
registry: ${{ env.REGISTRY }}
|
registry: ${{ env.REGISTRY }}
|
||||||
registry_username: ${{ github.actor }}
|
registry_username: ${{ github.actor }}
|
||||||
|
|
|
||||||
2
.github/workflows/issue-keyword-labeler.yml
vendored
2
.github/workflows/issue-keyword-labeler.yml
vendored
|
|
@ -19,7 +19,7 @@ jobs:
|
||||||
id: scan
|
id: scan
|
||||||
env:
|
env:
|
||||||
PROVIDER_ISSUE_WEBHOOK_URL: ${{ secrets.PROVIDER_ISSUE_WEBHOOK_URL }}
|
PROVIDER_ISSUE_WEBHOOK_URL: ${{ secrets.PROVIDER_ISSUE_WEBHOOK_URL }}
|
||||||
KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic
|
KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic,gemini,cohere,mistral,groq,ollama,deepseek
|
||||||
run: python3 .github/scripts/scan_keywords.py
|
run: python3 .github/scripts/scan_keywords.py
|
||||||
|
|
||||||
- name: Ensure label exists
|
- name: Ensure label exists
|
||||||
|
|
|
||||||
116
.github/workflows/label-component.yml
vendored
Normal file
116
.github/workflows/label-component.yml
vendored
Normal file
|
|
@ -0,0 +1,116 @@
|
||||||
|
name: Label Component Issues
|
||||||
|
|
||||||
|
on:
|
||||||
|
issues:
|
||||||
|
types:
|
||||||
|
- opened
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
add-component-label:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
issues: write
|
||||||
|
steps:
|
||||||
|
- name: Add component labels
|
||||||
|
uses: actions/github-script@v7
|
||||||
|
with:
|
||||||
|
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
script: |
|
||||||
|
const body = context.payload.issue.body;
|
||||||
|
if (!body) return;
|
||||||
|
|
||||||
|
// Define component mappings with regex patterns that handle flexible whitespace
|
||||||
|
const components = [
|
||||||
|
{
|
||||||
|
pattern: /What part of LiteLLM is this about\?\s*SDK \(litellm Python package\)/,
|
||||||
|
label: 'sdk',
|
||||||
|
color: '0E7C86',
|
||||||
|
description: 'Issues related to the litellm Python SDK'
|
||||||
|
},
|
||||||
|
{
|
||||||
|
pattern: /What part of LiteLLM is this about\?\s*Proxy/,
|
||||||
|
label: 'proxy',
|
||||||
|
color: '5319E7',
|
||||||
|
description: 'Issues related to the LiteLLM Proxy'
|
||||||
|
},
|
||||||
|
{
|
||||||
|
pattern: /What part of LiteLLM is this about\?\s*UI Dashboard/,
|
||||||
|
label: 'ui-dashboard',
|
||||||
|
color: 'D876E3',
|
||||||
|
description: 'Issues related to the LiteLLM UI Dashboard'
|
||||||
|
},
|
||||||
|
{
|
||||||
|
pattern: /What part of LiteLLM is this about\?\s*Docs/,
|
||||||
|
label: 'docs',
|
||||||
|
color: 'FBCA04',
|
||||||
|
description: 'Issues related to LiteLLM documentation'
|
||||||
|
}
|
||||||
|
];
|
||||||
|
|
||||||
|
// Find matching component
|
||||||
|
for (const component of components) {
|
||||||
|
if (component.pattern.test(body)) {
|
||||||
|
// Ensure label exists
|
||||||
|
try {
|
||||||
|
await github.rest.issues.getLabel({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
name: component.label
|
||||||
|
});
|
||||||
|
} catch (error) {
|
||||||
|
if (error.status === 404) {
|
||||||
|
await github.rest.issues.createLabel({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
name: component.label,
|
||||||
|
color: component.color,
|
||||||
|
description: component.description
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Add label to issue
|
||||||
|
await github.rest.issues.addLabels({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
issue_number: context.issue.number,
|
||||||
|
labels: [component.label]
|
||||||
|
});
|
||||||
|
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check for 'claude code' keyword (can be applied alongside component labels)
|
||||||
|
if (/claude code/i.test(body)) {
|
||||||
|
const claudeLabel = {
|
||||||
|
name: 'claude code',
|
||||||
|
color: '7c3aed',
|
||||||
|
description: 'Issues related to Claude Code usage'
|
||||||
|
};
|
||||||
|
|
||||||
|
try {
|
||||||
|
await github.rest.issues.getLabel({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
name: claudeLabel.name
|
||||||
|
});
|
||||||
|
} catch (error) {
|
||||||
|
if (error.status === 404) {
|
||||||
|
await github.rest.issues.createLabel({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
name: claudeLabel.name,
|
||||||
|
color: claudeLabel.color,
|
||||||
|
description: claudeLabel.description
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
await github.rest.issues.addLabels({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
issue_number: context.issue.number,
|
||||||
|
labels: [claudeLabel.name]
|
||||||
|
});
|
||||||
|
}
|
||||||
17
.github/workflows/label-mlops.yml
vendored
17
.github/workflows/label-mlops.yml
vendored
|
|
@ -1,17 +0,0 @@
|
||||||
name: Label ML Ops Team Issues
|
|
||||||
|
|
||||||
on:
|
|
||||||
issues:
|
|
||||||
types:
|
|
||||||
- opened
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
add-mlops-label:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- name: Check if ML Ops Team is selected
|
|
||||||
uses: actions-ecosystem/action-add-labels@v1
|
|
||||||
if: contains(github.event.issue.body, '### Are you a ML Ops Team?') && contains(github.event.issue.body, 'Yes')
|
|
||||||
with:
|
|
||||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
|
||||||
labels: "mlops user request"
|
|
||||||
1
.github/workflows/publish-migrations.yml
vendored
1
.github/workflows/publish-migrations.yml
vendored
|
|
@ -13,6 +13,7 @@ on:
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
publish-migrations:
|
publish-migrations:
|
||||||
|
if: github.repository == 'BerriAI/litellm'
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
services:
|
services:
|
||||||
postgres:
|
postgres:
|
||||||
|
|
|
||||||
2
.github/workflows/test-linting.yml
vendored
2
.github/workflows/test-linting.yml
vendored
|
|
@ -73,4 +73,4 @@ jobs:
|
||||||
|
|
||||||
- name: Check import safety
|
- name: Check import safety
|
||||||
run: |
|
run: |
|
||||||
poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
|
poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
|
||||||
|
|
|
||||||
95
.github/workflows/test-litellm-matrix.yml
vendored
Normal file
95
.github/workflows/test-litellm-matrix.yml
vendored
Normal file
|
|
@ -0,0 +1,95 @@
|
||||||
|
name: LiteLLM Unit Tests (Matrix)
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches: [main]
|
||||||
|
|
||||||
|
# Cancel in-progress runs for the same PR
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
timeout-minutes: 15
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
test-group:
|
||||||
|
# tests/test_litellm split by subdirectory (~560 files total)
|
||||||
|
- name: "llms"
|
||||||
|
path: "tests/test_litellm/llms"
|
||||||
|
workers: 4
|
||||||
|
# tests/test_litellm/proxy split by subdirectory (~180 files total)
|
||||||
|
- name: "proxy-guardrails"
|
||||||
|
path: "tests/test_litellm/proxy/guardrails tests/test_litellm/proxy/management_endpoints tests/test_litellm/proxy/management_helpers"
|
||||||
|
workers: 4
|
||||||
|
- name: "proxy-core"
|
||||||
|
path: "tests/test_litellm/proxy/auth tests/test_litellm/proxy/client tests/test_litellm/proxy/db tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine"
|
||||||
|
workers: 4
|
||||||
|
- name: "proxy-misc"
|
||||||
|
path: "tests/test_litellm/proxy/_experimental tests/test_litellm/proxy/agent_endpoints tests/test_litellm/proxy/anthropic_endpoints tests/test_litellm/proxy/common_utils tests/test_litellm/proxy/discovery_endpoints tests/test_litellm/proxy/experimental tests/test_litellm/proxy/google_endpoints tests/test_litellm/proxy/health_endpoints tests/test_litellm/proxy/image_endpoints tests/test_litellm/proxy/middleware tests/test_litellm/proxy/openai_files_endpoint tests/test_litellm/proxy/pass_through_endpoints tests/test_litellm/proxy/prompts tests/test_litellm/proxy/public_endpoints tests/test_litellm/proxy/response_api_endpoints tests/test_litellm/proxy/spend_tracking tests/test_litellm/proxy/ui_crud_endpoints tests/test_litellm/proxy/vector_store_endpoints tests/test_litellm/proxy/test_*.py"
|
||||||
|
workers: 4
|
||||||
|
- name: "integrations"
|
||||||
|
path: "tests/test_litellm/integrations"
|
||||||
|
workers: 4
|
||||||
|
- name: "core-utils"
|
||||||
|
path: "tests/test_litellm/litellm_core_utils"
|
||||||
|
workers: 2
|
||||||
|
- name: "other"
|
||||||
|
path: "tests/test_litellm/caching tests/test_litellm/responses tests/test_litellm/secret_managers tests/test_litellm/vector_stores tests/test_litellm/a2a_protocol tests/test_litellm/anthropic_interface tests/test_litellm/completion_extras tests/test_litellm/containers tests/test_litellm/enterprise tests/test_litellm/experimental_mcp_client tests/test_litellm/google_genai tests/test_litellm/images tests/test_litellm/interactions tests/test_litellm/passthrough tests/test_litellm/router_strategy tests/test_litellm/router_utils tests/test_litellm/types"
|
||||||
|
workers: 4
|
||||||
|
- name: "root"
|
||||||
|
path: "tests/test_litellm/test_*.py"
|
||||||
|
workers: 4
|
||||||
|
# tests/proxy_unit_tests split alphabetically (~48 files total)
|
||||||
|
- name: "proxy-unit-a"
|
||||||
|
path: "tests/proxy_unit_tests/test_[a-o]*.py"
|
||||||
|
workers: 2
|
||||||
|
- name: "proxy-unit-b"
|
||||||
|
path: "tests/proxy_unit_tests/test_[p-z]*.py"
|
||||||
|
workers: 2
|
||||||
|
|
||||||
|
name: test (${{ matrix.test-group.name }})
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.12"
|
||||||
|
|
||||||
|
- name: Install Poetry
|
||||||
|
uses: snok/install-poetry@v1
|
||||||
|
|
||||||
|
- name: Cache Poetry dependencies
|
||||||
|
uses: actions/cache@v4
|
||||||
|
with:
|
||||||
|
path: |
|
||||||
|
~/.cache/pypoetry
|
||||||
|
~/.cache/pip
|
||||||
|
.venv
|
||||||
|
key: ${{ runner.os }}-poetry-${{ hashFiles('poetry.lock') }}
|
||||||
|
restore-keys: |
|
||||||
|
${{ runner.os }}-poetry-
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: |
|
||||||
|
poetry config virtualenvs.in-project true
|
||||||
|
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
|
||||||
|
poetry run pip install pytest-retry==1.6.3 pytest-xdist google-genai==1.22.0 \
|
||||||
|
google-cloud-aiplatform>=1.38 fastapi-offline==1.7.3 python-multipart==0.0.22 openapi-core
|
||||||
|
|
||||||
|
- name: Setup litellm-enterprise
|
||||||
|
run: |
|
||||||
|
cd enterprise && poetry run pip install -e . && cd ..
|
||||||
|
|
||||||
|
- name: Run tests - ${{ matrix.test-group.name }}
|
||||||
|
run: |
|
||||||
|
poetry run pytest ${{ matrix.test-group.path }} \
|
||||||
|
--tb=short -vv \
|
||||||
|
--maxfail=10 \
|
||||||
|
-n ${{ matrix.test-group.workers }} \
|
||||||
|
--durations=20
|
||||||
32
.github/workflows/test-litellm-ui-build.yml
vendored
Normal file
32
.github/workflows/test-litellm-ui-build.yml
vendored
Normal file
|
|
@ -0,0 +1,32 @@
|
||||||
|
name: UI Build Check
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches: [main]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build-ui:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
timeout-minutes: 10
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
working-directory: ui/litellm-dashboard
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Setup Node.js
|
||||||
|
uses: actions/setup-node@v4
|
||||||
|
with:
|
||||||
|
node-version: "20"
|
||||||
|
cache: "npm"
|
||||||
|
cache-dependency-path: ui/litellm-dashboard/package-lock.json
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: npm install
|
||||||
|
|
||||||
|
- name: Build
|
||||||
|
run: npm run build
|
||||||
11
.github/workflows/test-litellm.yml
vendored
11
.github/workflows/test-litellm.yml
vendored
|
|
@ -1,8 +1,12 @@
|
||||||
name: LiteLLM Mock Tests (folder - tests/test_litellm)
|
name: LiteLLM Mock Tests (folder - tests/test_litellm)
|
||||||
|
|
||||||
|
# DEPRECATED: This workflow is replaced by test-litellm-matrix.yml which runs
|
||||||
|
# the same tests in parallel across 10 jobs for faster CI times.
|
||||||
|
# Kept for manual debugging only.
|
||||||
on:
|
on:
|
||||||
pull_request:
|
workflow_dispatch: # Manual trigger only
|
||||||
branches: [ main ]
|
# pull_request:
|
||||||
|
# branches: [ main ]
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
test:
|
test:
|
||||||
|
|
@ -34,7 +38,8 @@ jobs:
|
||||||
poetry run pip install "google-genai==1.22.0"
|
poetry run pip install "google-genai==1.22.0"
|
||||||
poetry run pip install "google-cloud-aiplatform>=1.38"
|
poetry run pip install "google-cloud-aiplatform>=1.38"
|
||||||
poetry run pip install "fastapi-offline==1.7.3"
|
poetry run pip install "fastapi-offline==1.7.3"
|
||||||
poetry run pip install "python-multipart==0.0.18"
|
poetry run pip install "python-multipart==0.0.22"
|
||||||
|
poetry run pip install "openapi-core"
|
||||||
- name: Setup litellm-enterprise as local package
|
- name: Setup litellm-enterprise as local package
|
||||||
run: |
|
run: |
|
||||||
cd enterprise
|
cd enterprise
|
||||||
|
|
|
||||||
4
.github/workflows/test-mcp.yml
vendored
4
.github/workflows/test-mcp.yml
vendored
|
|
@ -34,8 +34,8 @@ jobs:
|
||||||
poetry run pip install "pytest-cov==5.0.0"
|
poetry run pip install "pytest-cov==5.0.0"
|
||||||
poetry run pip install "pytest-asyncio==0.21.1"
|
poetry run pip install "pytest-asyncio==0.21.1"
|
||||||
poetry run pip install "respx==0.22.0"
|
poetry run pip install "respx==0.22.0"
|
||||||
poetry run pip install "pydantic==2.10.2"
|
poetry run pip install "pydantic==2.11.0"
|
||||||
poetry run pip install "mcp==1.10.1"
|
poetry run pip install "mcp==1.25.0"
|
||||||
poetry run pip install pytest-xdist
|
poetry run pip install pytest-xdist
|
||||||
|
|
||||||
- name: Setup litellm-enterprise as local package
|
- name: Setup litellm-enterprise as local package
|
||||||
|
|
|
||||||
15
.github/workflows/test-model-map.yaml
vendored
Normal file
15
.github/workflows/test-model-map.yaml
vendored
Normal file
|
|
@ -0,0 +1,15 @@
|
||||||
|
name: Validate model_prices_and_context_window.json
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches: [ main ]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
validate-model-prices-json:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Validate model_prices_and_context_window.json
|
||||||
|
run: |
|
||||||
|
jq empty model_prices_and_context_window.json
|
||||||
96
.github/workflows/test_server_root_path.yml
vendored
Normal file
96
.github/workflows/test_server_root_path.yml
vendored
Normal file
|
|
@ -0,0 +1,96 @@
|
||||||
|
name: Test Proxy SERVER_ROOT_PATH Routing
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches: [main]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test-server-root-path:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
timeout-minutes: 15
|
||||||
|
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
root_path: ["/api/v1", "/llmproxy"]
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
|
- name: Build Docker image
|
||||||
|
uses: docker/build-push-action@v5
|
||||||
|
with:
|
||||||
|
context: .
|
||||||
|
file: ./docker/Dockerfile.database
|
||||||
|
tags: litellm-test:${{ github.sha }}
|
||||||
|
load: true
|
||||||
|
cache-from: type=gha
|
||||||
|
cache-to: type=gha,mode=max
|
||||||
|
|
||||||
|
- name: Start LiteLLM container with SERVER_ROOT_PATH
|
||||||
|
run: |
|
||||||
|
docker run -d \
|
||||||
|
--name litellm-test \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e SERVER_ROOT_PATH="${{ matrix.root_path }}" \
|
||||||
|
-e LITELLM_MASTER_KEY="sk-1234" \
|
||||||
|
litellm-test:${{ github.sha }} \
|
||||||
|
--detailed_debug
|
||||||
|
|
||||||
|
- name: Wait for container to be healthy
|
||||||
|
run: |
|
||||||
|
echo "Waiting for LiteLLM to start..."
|
||||||
|
max_attempts=30
|
||||||
|
attempt=0
|
||||||
|
|
||||||
|
while [ $attempt -lt $max_attempts ]; do
|
||||||
|
if docker logs litellm-test 2>&1 | grep -q "Uvicorn running"; then
|
||||||
|
echo "LiteLLM started successfully"
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
attempt=$((attempt + 1))
|
||||||
|
echo "Attempt $attempt/$max_attempts - waiting for server to start..."
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
|
||||||
|
if [ $attempt -eq $max_attempts ]; then
|
||||||
|
echo "Server failed to start within timeout"
|
||||||
|
docker logs litellm-test
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 5
|
||||||
|
|
||||||
|
- name: Show container logs
|
||||||
|
if: always()
|
||||||
|
run: docker logs litellm-test
|
||||||
|
|
||||||
|
- name: Test UI endpoint with root path
|
||||||
|
run: |
|
||||||
|
ROOT_PATH="${{ matrix.root_path }}"
|
||||||
|
echo "Testing UI at: http://localhost:4000${ROOT_PATH}/ui/"
|
||||||
|
|
||||||
|
for i in 1 2 3; do
|
||||||
|
content=$(curl -sL --max-time 5 -H "Authorization: Bearer sk-1234" "http://localhost:4000${ROOT_PATH}/ui/")
|
||||||
|
if echo "$content" | grep -q -E "(html|<!DOCTYPE|<head|<body)"; then
|
||||||
|
echo "UI page contains valid HTML content"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
echo "Attempt $i/3 - no valid HTML, retrying in 5s..."
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
echo "UI page does not contain expected HTML content"
|
||||||
|
echo "Response: $content"
|
||||||
|
docker logs litellm-test
|
||||||
|
exit 1
|
||||||
|
|
||||||
|
- name: Cleanup
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
docker stop litellm-test || true
|
||||||
|
docker rm litellm-test || true
|
||||||
15
.gitignore
vendored
15
.gitignore
vendored
|
|
@ -1,6 +1,8 @@
|
||||||
.python-version
|
.python-version
|
||||||
.venv
|
.venv
|
||||||
|
.venv_policy_test
|
||||||
.env
|
.env
|
||||||
|
.claude
|
||||||
.newenv
|
.newenv
|
||||||
newenv/*
|
newenv/*
|
||||||
litellm/proxy/myenv/*
|
litellm/proxy/myenv/*
|
||||||
|
|
@ -59,9 +61,6 @@ litellm/proxy/_super_secret_config.yaml
|
||||||
litellm/proxy/myenv/bin/activate
|
litellm/proxy/myenv/bin/activate
|
||||||
litellm/proxy/myenv/bin/Activate.ps1
|
litellm/proxy/myenv/bin/Activate.ps1
|
||||||
myenv/*
|
myenv/*
|
||||||
litellm/proxy/_experimental/out/404/index.html
|
|
||||||
litellm/proxy/_experimental/out/model_hub/index.html
|
|
||||||
litellm/proxy/_experimental/out/onboarding/index.html
|
|
||||||
litellm/tests/log.txt
|
litellm/tests/log.txt
|
||||||
litellm/tests/langfuse.log
|
litellm/tests/langfuse.log
|
||||||
litellm/tests/langfuse.log
|
litellm/tests/langfuse.log
|
||||||
|
|
@ -74,9 +73,6 @@ tests/local_testing/log.txt
|
||||||
litellm/proxy/_new_new_secret_config.yaml
|
litellm/proxy/_new_new_secret_config.yaml
|
||||||
litellm/proxy/custom_guardrail.py
|
litellm/proxy/custom_guardrail.py
|
||||||
.mypy_cache/*
|
.mypy_cache/*
|
||||||
litellm/proxy/_experimental/out/404.html
|
|
||||||
litellm/proxy/_experimental/out/404.html
|
|
||||||
litellm/proxy/_experimental/out/model_hub.html
|
|
||||||
.mypy_cache/*
|
.mypy_cache/*
|
||||||
litellm/proxy/application.log
|
litellm/proxy/application.log
|
||||||
tests/llm_translation/vertex_test_account.json
|
tests/llm_translation/vertex_test_account.json
|
||||||
|
|
@ -98,5 +94,10 @@ litellm_config.yaml
|
||||||
litellm/proxy/to_delete_loadtest_work/*
|
litellm/proxy/to_delete_loadtest_work/*
|
||||||
update_model_cost_map.py
|
update_model_cost_map.py
|
||||||
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
|
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
|
||||||
litellm/proxy/_experimental/out/guardrails/index.html
|
|
||||||
scripts/test_vertex_ai_search.py
|
scripts/test_vertex_ai_search.py
|
||||||
|
LAZY_LOADING_IMPROVEMENTS.md
|
||||||
|
STABILIZATION_TODO.md
|
||||||
|
**/test-results
|
||||||
|
**/playwright-report
|
||||||
|
**/*.storageState.json
|
||||||
|
**/coverage
|
||||||
22
.semgrep/rules/README.md
Normal file
22
.semgrep/rules/README.md
Normal file
|
|
@ -0,0 +1,22 @@
|
||||||
|
# Custom Semgrep rules for LiteLLM
|
||||||
|
|
||||||
|
Add custom rule YAML files here. Semgrep loads all `.yml`/`.yaml` files under this directory.
|
||||||
|
|
||||||
|
**Run only custom rules (CI / fail on findings):**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
semgrep scan --config .semgrep/rules . --error
|
||||||
|
```
|
||||||
|
|
||||||
|
**Run with registry + custom rules:**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
semgrep scan --config auto --config .semgrep/rules .
|
||||||
|
```
|
||||||
|
|
||||||
|
**Layout:**
|
||||||
|
|
||||||
|
- `python/` – Python-specific rules (security, patterns)
|
||||||
|
- Add more subdirs as needed (e.g. `generic/` for language-agnostic rules)
|
||||||
|
|
||||||
|
See [Semgrep rule syntax](https://semgrep.dev/docs/writing-rules/rule-syntax/).
|
||||||
17
.semgrep/rules/python/reliability/unbounded-memory.yml
Normal file
17
.semgrep/rules/python/reliability/unbounded-memory.yml
Normal file
|
|
@ -0,0 +1,17 @@
|
||||||
|
# Unbounded memory growth – data structures without a clear max limit
|
||||||
|
# Can lead to OOM under load.
|
||||||
|
|
||||||
|
rules:
|
||||||
|
- id: unbounded-asyncio-queue
|
||||||
|
message: asyncio.Queue() with no maxsize can grow unbounded. Use asyncio.Queue(maxsize=N) for integrations (e.g. log queues).
|
||||||
|
severity: ERROR
|
||||||
|
languages: [python]
|
||||||
|
pattern-either:
|
||||||
|
- pattern: asyncio.Queue()
|
||||||
|
- pattern: asyncio.Queue(maxsize=0)
|
||||||
|
metadata:
|
||||||
|
category: reliability
|
||||||
|
cwe: "CWE-400: Uncontrolled Resource Consumption"
|
||||||
|
tags: [python, reliability]
|
||||||
|
confidence: HIGH
|
||||||
|
source: https://docs.python.org/3/library/asyncio-queue.html
|
||||||
14
.semgrep/rules/python/unbounded-memory.yml
Normal file
14
.semgrep/rules/python/unbounded-memory.yml
Normal file
|
|
@ -0,0 +1,14 @@
|
||||||
|
# Unbounded memory growth – data structures without a clear max limit
|
||||||
|
# Can lead to OOM under load.
|
||||||
|
|
||||||
|
rules:
|
||||||
|
- id: unbounded-asyncio-queue
|
||||||
|
message: asyncio.Queue() with no maxsize can grow unbounded. Use asyncio.Queue(maxsize=N) for integrations (e.g. log queues).
|
||||||
|
severity: ERROR
|
||||||
|
languages: [python]
|
||||||
|
pattern-either:
|
||||||
|
- pattern: asyncio.Queue()
|
||||||
|
- pattern: asyncio.Queue(maxsize=0)
|
||||||
|
metadata:
|
||||||
|
category: correctness
|
||||||
|
cwe: "CWE-400: Uncontrolled Resource Consumption"
|
||||||
12
.trivyignore
Normal file
12
.trivyignore
Normal file
|
|
@ -0,0 +1,12 @@
|
||||||
|
# LiteLLM Trivy Ignore File
|
||||||
|
# CVEs listed here are temporarily allowlisted pending fixes
|
||||||
|
|
||||||
|
# Next.js vulnerabilities in UI dashboard (next@14.2.35)
|
||||||
|
# Allowlisted: 2026-01-31, 7-day fix timeline
|
||||||
|
# Fix: Upgrade to Next.js 15.5.10+ or 16.1.5+
|
||||||
|
|
||||||
|
# HIGH: DoS via request deserialization
|
||||||
|
GHSA-h25m-26qc-wcjf
|
||||||
|
|
||||||
|
# MEDIUM: Image Optimizer DoS
|
||||||
|
CVE-2025-59471
|
||||||
23
AGENTS.md
23
AGENTS.md
|
|
@ -49,6 +49,29 @@ LiteLLM is a unified interface for 100+ LLMs that:
|
||||||
- Test provider-specific functionality thoroughly
|
- Test provider-specific functionality thoroughly
|
||||||
- Consider adding load tests for performance-critical changes
|
- Consider adding load tests for performance-critical changes
|
||||||
|
|
||||||
|
### MAKING CODE CHANGES FOR THE UI (IGNORE FOR BACKEND)
|
||||||
|
|
||||||
|
1. **Tremor is DEPRECATED, do not use Tremor components in new features/changes**
|
||||||
|
- The only exception is the Tremor Table component and its required Tremor Table sub components.
|
||||||
|
|
||||||
|
2. **Use Common Components as much as possible**:
|
||||||
|
- These are usually defined in the `common_components` directory
|
||||||
|
- Use these components as much as possible and avoid building new components unless needed
|
||||||
|
|
||||||
|
3. **Testing**:
|
||||||
|
- The codebase uses **Vitest** and **React Testing Library**
|
||||||
|
- **Query Priority Order**: Use query methods in this order: `getByRole`, `getByLabelText`, `getByPlaceholderText`, `getByText`, `getByTestId`
|
||||||
|
- **Always use `screen`** instead of destructuring from `render()` (e.g., use `screen.getByText()` not `getByText`)
|
||||||
|
- **Wrap user interactions in `act()`**: Always wrap `fireEvent` calls with `act()` to ensure React state updates are properly handled
|
||||||
|
- **Use `query` methods for absence checks**: Use `queryBy*` methods (not `getBy*`) when expecting an element to NOT be present
|
||||||
|
- **Test names must start with "should"**: All test names should follow the pattern `it("should ...")`
|
||||||
|
- **Mock external dependencies**: Check `setupTests.ts` for global mocks and mock child components/networking calls as needed
|
||||||
|
- **Structure tests properly**:
|
||||||
|
- First test should verify the component renders successfully
|
||||||
|
- Subsequent tests should focus on functionality and user interactions
|
||||||
|
- Use `waitFor` for async operations that aren't already awaited
|
||||||
|
- **Avoid using `querySelector`**: Prefer React Testing Library queries over direct DOM manipulation
|
||||||
|
|
||||||
### IMPORTANT PATTERNS
|
### IMPORTANT PATTERNS
|
||||||
|
|
||||||
1. **Function/Tool Calling**:
|
1. **Function/Tool Calling**:
|
||||||
|
|
|
||||||
398
ARCHITECTURE.md
Normal file
398
ARCHITECTURE.md
Normal file
|
|
@ -0,0 +1,398 @@
|
||||||
|
# LiteLLM Architecture - LiteLLM SDK + AI Gateway
|
||||||
|
|
||||||
|
This document helps contributors understand where to make changes in LiteLLM.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
The LiteLLM AI Gateway (Proxy) uses the LiteLLM SDK internally for all LLM calls:
|
||||||
|
|
||||||
|
```
|
||||||
|
OpenAI SDK (client) ──▶ LiteLLM AI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
|
||||||
|
Anthropic SDK (client) ──▶ LiteLLMAI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
|
||||||
|
Any HTTP client ──▶ LiteLLMAI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
|
||||||
|
```
|
||||||
|
|
||||||
|
The **AI Gateway** adds authentication, rate limiting, budgets, and routing on top of the SDK.
|
||||||
|
The **SDK** handles the actual LLM provider calls, request/response transformations, and streaming.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. AI Gateway (Proxy) Request Flow
|
||||||
|
|
||||||
|
The AI Gateway (`litellm/proxy/`) wraps the SDK with authentication, rate limiting, and management features.
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
sequenceDiagram
|
||||||
|
participant Client
|
||||||
|
participant ProxyServer as proxy/proxy_server.py
|
||||||
|
participant Auth as proxy/auth/user_api_key_auth.py
|
||||||
|
participant Redis as Redis Cache
|
||||||
|
participant Hooks as proxy/hooks/
|
||||||
|
participant Router as router.py
|
||||||
|
participant Main as main.py + utils.py
|
||||||
|
participant Handler as llms/custom_httpx/llm_http_handler.py
|
||||||
|
participant Transform as llms/{provider}/chat/transformation.py
|
||||||
|
participant Provider as LLM Provider API
|
||||||
|
participant CostCalc as cost_calculator.py
|
||||||
|
participant LoggingObj as litellm_logging.py
|
||||||
|
participant DBWriter as db/db_spend_update_writer.py
|
||||||
|
participant Postgres as PostgreSQL
|
||||||
|
|
||||||
|
%% Request Flow
|
||||||
|
Client->>ProxyServer: POST /v1/chat/completions
|
||||||
|
ProxyServer->>Auth: user_api_key_auth()
|
||||||
|
Auth->>Redis: Check API key cache
|
||||||
|
Redis-->>Auth: Key info + spend limits
|
||||||
|
ProxyServer->>Hooks: max_budget_limiter, parallel_request_limiter
|
||||||
|
Hooks->>Redis: Check/increment rate limit counters
|
||||||
|
ProxyServer->>Router: route_request()
|
||||||
|
Router->>Main: litellm.acompletion()
|
||||||
|
Main->>Handler: BaseLLMHTTPHandler.completion()
|
||||||
|
Handler->>Transform: ProviderConfig.transform_request()
|
||||||
|
Handler->>Provider: HTTP Request
|
||||||
|
Provider-->>Handler: Response
|
||||||
|
Handler->>Transform: ProviderConfig.transform_response()
|
||||||
|
Transform-->>Handler: ModelResponse
|
||||||
|
Handler-->>Main: ModelResponse
|
||||||
|
|
||||||
|
%% Cost Attribution (in utils.py wrapper)
|
||||||
|
Main->>LoggingObj: update_response_metadata()
|
||||||
|
LoggingObj->>CostCalc: _response_cost_calculator()
|
||||||
|
CostCalc->>CostCalc: completion_cost(tokens × price)
|
||||||
|
CostCalc-->>LoggingObj: response_cost
|
||||||
|
LoggingObj-->>Main: Set response._hidden_params["response_cost"]
|
||||||
|
Main-->>ProxyServer: ModelResponse (with cost in _hidden_params)
|
||||||
|
|
||||||
|
%% Response Headers + Async Logging
|
||||||
|
ProxyServer->>ProxyServer: Extract cost from hidden_params
|
||||||
|
ProxyServer->>LoggingObj: async_success_handler()
|
||||||
|
LoggingObj->>Hooks: async_log_success_event()
|
||||||
|
Hooks->>DBWriter: update_database(response_cost)
|
||||||
|
DBWriter->>Redis: Queue spend increment
|
||||||
|
DBWriter->>Postgres: Batch write spend logs (async)
|
||||||
|
ProxyServer-->>Client: ModelResponse + x-litellm-response-cost header
|
||||||
|
```
|
||||||
|
|
||||||
|
### Proxy Components
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
graph TD
|
||||||
|
subgraph "Incoming Request"
|
||||||
|
Client["POST /v1/chat/completions"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "proxy/proxy_server.py"
|
||||||
|
Endpoint["chat_completion()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "proxy/auth/"
|
||||||
|
Auth["user_api_key_auth()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "proxy/"
|
||||||
|
PreCall["litellm_pre_call_utils.py"]
|
||||||
|
RouteRequest["route_llm_request.py"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "litellm/"
|
||||||
|
Router["router.py"]
|
||||||
|
Main["main.py"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "Infrastructure"
|
||||||
|
DualCache["DualCache<br/>(in-memory + Redis)"]
|
||||||
|
Postgres["PostgreSQL<br/>(keys, teams, spend logs)"]
|
||||||
|
end
|
||||||
|
|
||||||
|
Client --> Endpoint
|
||||||
|
Endpoint --> Auth
|
||||||
|
Auth --> DualCache
|
||||||
|
DualCache -.->|cache miss| Postgres
|
||||||
|
Auth --> PreCall
|
||||||
|
PreCall --> RouteRequest
|
||||||
|
RouteRequest --> Router
|
||||||
|
Router --> DualCache
|
||||||
|
Router --> Main
|
||||||
|
Main --> Client
|
||||||
|
```
|
||||||
|
|
||||||
|
**Key proxy files:**
|
||||||
|
- `proxy/proxy_server.py` - Main API endpoints
|
||||||
|
- `proxy/auth/` - Authentication (API keys, JWT, OAuth2)
|
||||||
|
- `proxy/hooks/` - Proxy-level callbacks
|
||||||
|
- `router.py` - Load balancing, fallbacks
|
||||||
|
- `router_strategy/` - Routing algorithms (`lowest_latency.py`, `simple_shuffle.py`, etc.)
|
||||||
|
|
||||||
|
**LLM-specific proxy endpoints:**
|
||||||
|
|
||||||
|
| Endpoint | Directory | Purpose |
|
||||||
|
|----------|-----------|---------|
|
||||||
|
| `/v1/messages` | `proxy/anthropic_endpoints/` | Anthropic Messages API |
|
||||||
|
| `/vertex-ai/*` | `proxy/vertex_ai_endpoints/` | Vertex AI passthrough |
|
||||||
|
| `/gemini/*` | `proxy/google_endpoints/` | Google AI Studio passthrough |
|
||||||
|
| `/v1/images/*` | `proxy/image_endpoints/` | Image generation |
|
||||||
|
| `/v1/batches` | `proxy/batches_endpoints/` | Batch processing |
|
||||||
|
| `/v1/files` | `proxy/openai_files_endpoints/` | File uploads |
|
||||||
|
| `/v1/fine_tuning` | `proxy/fine_tuning_endpoints/` | Fine-tuning jobs |
|
||||||
|
| `/v1/rerank` | `proxy/rerank_endpoints/` | Reranking |
|
||||||
|
| `/v1/responses` | `proxy/response_api_endpoints/` | OpenAI Responses API |
|
||||||
|
| `/v1/vector_stores` | `proxy/vector_store_endpoints/` | Vector stores |
|
||||||
|
| `/*` (passthrough) | `proxy/pass_through_endpoints/` | Direct provider passthrough |
|
||||||
|
|
||||||
|
**Proxy Hooks** (`proxy/hooks/__init__.py`):
|
||||||
|
|
||||||
|
| Hook | File | Purpose |
|
||||||
|
|------|------|---------|
|
||||||
|
| `max_budget_limiter` | `proxy/hooks/max_budget_limiter.py` | Enforce budget limits |
|
||||||
|
| `parallel_request_limiter` | `proxy/hooks/parallel_request_limiter_v3.py` | Rate limiting per key/user |
|
||||||
|
| `cache_control_check` | `proxy/hooks/cache_control_check.py` | Cache validation |
|
||||||
|
| `responses_id_security` | `proxy/hooks/responses_id_security.py` | Response ID validation |
|
||||||
|
| `litellm_skills` | `proxy/hooks/skills_injection.py` | Skills injection |
|
||||||
|
|
||||||
|
To add a new proxy hook, implement `CustomLogger` and register in `PROXY_HOOKS`.
|
||||||
|
|
||||||
|
### Infrastructure Components
|
||||||
|
|
||||||
|
The AI Gateway uses external infrastructure for persistence and caching:
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
graph LR
|
||||||
|
subgraph "AI Gateway (proxy/)"
|
||||||
|
Proxy["proxy_server.py"]
|
||||||
|
Auth["auth/user_api_key_auth.py"]
|
||||||
|
DBWriter["db/db_spend_update_writer.py<br/>DBSpendUpdateWriter"]
|
||||||
|
InternalCache["utils.py<br/>InternalUsageCache"]
|
||||||
|
CostCallback["hooks/proxy_track_cost_callback.py<br/>_ProxyDBLogger"]
|
||||||
|
Scheduler["APScheduler<br/>ProxyStartupEvent"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "SDK (litellm/)"
|
||||||
|
Router["router.py<br/>Router.cache (DualCache)"]
|
||||||
|
LLMCache["caching/caching_handler.py<br/>LLMCachingHandler"]
|
||||||
|
CacheClass["caching/caching.py<br/>Cache"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "Redis (caching/redis_cache.py)"
|
||||||
|
RateLimit["Rate Limit Counters"]
|
||||||
|
SpendQueue["Spend Increment Queue"]
|
||||||
|
KeyCache["API Key Cache"]
|
||||||
|
TPM_RPM["TPM/RPM Tracking"]
|
||||||
|
Cooldowns["Deployment Cooldowns"]
|
||||||
|
LLMResponseCache["LLM Response Cache"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "PostgreSQL (proxy/schema.prisma)"
|
||||||
|
Keys["LiteLLM_VerificationToken"]
|
||||||
|
Teams["LiteLLM_TeamTable"]
|
||||||
|
SpendLogs["LiteLLM_SpendLogs"]
|
||||||
|
Users["LiteLLM_UserTable"]
|
||||||
|
end
|
||||||
|
|
||||||
|
Auth --> InternalCache
|
||||||
|
InternalCache --> KeyCache
|
||||||
|
InternalCache -.->|cache miss| Keys
|
||||||
|
InternalCache --> RateLimit
|
||||||
|
Router --> TPM_RPM
|
||||||
|
Router --> Cooldowns
|
||||||
|
LLMCache --> CacheClass
|
||||||
|
CacheClass --> LLMResponseCache
|
||||||
|
CostCallback --> DBWriter
|
||||||
|
DBWriter --> SpendQueue
|
||||||
|
DBWriter --> SpendLogs
|
||||||
|
Scheduler --> SpendLogs
|
||||||
|
Scheduler --> Keys
|
||||||
|
```
|
||||||
|
|
||||||
|
| Component | Purpose | Key Files/Classes |
|
||||||
|
|-----------|---------|-------------------|
|
||||||
|
| **Redis** | Rate limiting, API key caching, TPM/RPM tracking, cooldowns, LLM response caching, spend queuing | `caching/redis_cache.py` (`RedisCache`), `caching/dual_cache.py` (`DualCache`) |
|
||||||
|
| **PostgreSQL** | API keys, teams, users, spend logs | `proxy/utils.py` (`PrismaClient`), `proxy/schema.prisma` |
|
||||||
|
| **InternalUsageCache** | Proxy-level cache for rate limits + API keys (in-memory + Redis) | `proxy/utils.py` (`InternalUsageCache`) |
|
||||||
|
| **Router.cache** | TPM/RPM tracking, deployment cooldowns, client caching (in-memory + Redis) | `router.py` (`Router.cache: DualCache`) |
|
||||||
|
| **LLMCachingHandler** | SDK-level LLM response/embedding caching | `caching/caching_handler.py` (`LLMCachingHandler`), `caching/caching.py` (`Cache`) |
|
||||||
|
| **DBSpendUpdateWriter** | Batches spend updates to reduce DB writes | `proxy/db/db_spend_update_writer.py` (`DBSpendUpdateWriter`) |
|
||||||
|
| **Cost Tracking** | Calculates and logs response costs | `proxy/hooks/proxy_track_cost_callback.py` (`_ProxyDBLogger`) |
|
||||||
|
|
||||||
|
**Background Jobs** (APScheduler, initialized in `proxy/proxy_server.py` → `ProxyStartupEvent.initialize_scheduled_background_jobs()`):
|
||||||
|
|
||||||
|
| Job | Interval | Purpose | Key Files |
|
||||||
|
|-----|----------|---------|-----------|
|
||||||
|
| `update_spend` | 60s | Batch write spend logs to PostgreSQL | `proxy/db/db_spend_update_writer.py` |
|
||||||
|
| `reset_budget` | 10-12min | Reset budgets for keys/users/teams | `proxy/management_helpers/budget_reset_job.py` |
|
||||||
|
| `add_deployment` | 10s | Sync new model deployments from DB | `proxy/proxy_server.py` (`ProxyConfig`) |
|
||||||
|
| `cleanup_old_spend_logs` | cron/interval | Delete old spend logs | `proxy/management_helpers/spend_log_cleanup.py` |
|
||||||
|
| `check_batch_cost` | 30min | Calculate costs for batch jobs | `proxy/management_helpers/check_batch_cost_job.py` |
|
||||||
|
| `check_responses_cost` | 30min | Calculate costs for responses API | `proxy/management_helpers/check_responses_cost_job.py` |
|
||||||
|
| `process_rotations` | 1hr | Auto-rotate API keys | `proxy/management_helpers/key_rotation_manager.py` |
|
||||||
|
| `_run_background_health_check` | continuous | Health check model deployments | `proxy/proxy_server.py` |
|
||||||
|
| `send_weekly_spend_report` | weekly | Slack spend alerts | `proxy/utils.py` (`SlackAlerting`) |
|
||||||
|
| `send_monthly_spend_report` | monthly | Slack spend alerts | `proxy/utils.py` (`SlackAlerting`) |
|
||||||
|
|
||||||
|
**Cost Attribution Flow:**
|
||||||
|
1. LLM response returns to `utils.py` wrapper after `litellm.acompletion()` completes
|
||||||
|
2. `update_response_metadata()` (`llm_response_utils/response_metadata.py`) is called
|
||||||
|
3. `logging_obj._response_cost_calculator()` (`litellm_logging.py`) calculates cost via `litellm.completion_cost()` (`cost_calculator.py`)
|
||||||
|
4. Cost is stored in `response._hidden_params["response_cost"]`
|
||||||
|
5. `proxy/common_request_processing.py` extracts cost from `hidden_params` and adds to response headers (`x-litellm-response-cost`)
|
||||||
|
6. `logging_obj.async_success_handler()` triggers callbacks including `_ProxyDBLogger.async_log_success_event()`
|
||||||
|
7. `DBSpendUpdateWriter.update_database()` queues spend increments to Redis
|
||||||
|
8. Background job `update_spend` flushes queued spend to PostgreSQL every 60s
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. SDK Request Flow
|
||||||
|
|
||||||
|
The SDK (`litellm/`) provides the core LLM calling functionality used by both direct SDK users and the AI Gateway.
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
graph TD
|
||||||
|
subgraph "SDK Entry Points"
|
||||||
|
Completion["litellm.completion()"]
|
||||||
|
Messages["litellm.messages()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "main.py"
|
||||||
|
Main["completion()<br/>acompletion()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "utils.py"
|
||||||
|
GetProvider["get_llm_provider()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "llms/custom_httpx/"
|
||||||
|
Handler["llm_http_handler.py<br/>BaseLLMHTTPHandler"]
|
||||||
|
HTTP["http_handler.py<br/>HTTPHandler / AsyncHTTPHandler"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "llms/{provider}/chat/"
|
||||||
|
TransformReq["transform_request()"]
|
||||||
|
TransformResp["transform_response()"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "litellm_core_utils/"
|
||||||
|
Streaming["streaming_handler.py"]
|
||||||
|
end
|
||||||
|
|
||||||
|
subgraph "integrations/ (async, off main thread)"
|
||||||
|
Callbacks["custom_logger.py<br/>Langfuse, Datadog, etc."]
|
||||||
|
end
|
||||||
|
|
||||||
|
Completion --> Main
|
||||||
|
Messages --> Main
|
||||||
|
Main --> GetProvider
|
||||||
|
GetProvider --> Handler
|
||||||
|
Handler --> TransformReq
|
||||||
|
TransformReq --> HTTP
|
||||||
|
HTTP --> Provider["LLM Provider API"]
|
||||||
|
Provider --> HTTP
|
||||||
|
HTTP --> TransformResp
|
||||||
|
TransformResp --> Streaming
|
||||||
|
Streaming --> Response["ModelResponse"]
|
||||||
|
Response -.->|async| Callbacks
|
||||||
|
```
|
||||||
|
|
||||||
|
**Key SDK files:**
|
||||||
|
- `main.py` - Entry points: `completion()`, `acompletion()`, `embedding()`
|
||||||
|
- `utils.py` - `get_llm_provider()` resolves model → provider
|
||||||
|
- `llms/custom_httpx/llm_http_handler.py` - Central HTTP orchestrator
|
||||||
|
- `llms/custom_httpx/http_handler.py` - Low-level HTTP client
|
||||||
|
- `llms/{provider}/chat/transformation.py` - Provider-specific transformations
|
||||||
|
- `litellm_core_utils/streaming_handler.py` - Streaming response handling
|
||||||
|
- `integrations/` - Async callbacks (Langfuse, Datadog, etc.)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. Translation Layer
|
||||||
|
|
||||||
|
When a request comes in, it goes through a **translation layer** that converts between API formats.
|
||||||
|
Each translation is isolated in its own file, making it easy to test and modify independently.
|
||||||
|
|
||||||
|
### Where to find translations
|
||||||
|
|
||||||
|
| Incoming API | Provider | Translation File |
|
||||||
|
|--------------|----------|------------------|
|
||||||
|
| `/v1/chat/completions` | Anthropic | `llms/anthropic/chat/transformation.py` |
|
||||||
|
| `/v1/chat/completions` | Bedrock Converse | `llms/bedrock/chat/converse_transformation.py` |
|
||||||
|
| `/v1/chat/completions` | Bedrock Invoke | `llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py` |
|
||||||
|
| `/v1/chat/completions` | Gemini | `llms/gemini/chat/transformation.py` |
|
||||||
|
| `/v1/chat/completions` | Vertex AI | `llms/vertex_ai/gemini/transformation.py` |
|
||||||
|
| `/v1/chat/completions` | OpenAI | `llms/openai/chat/gpt_transformation.py` |
|
||||||
|
| `/v1/messages` (passthrough) | Anthropic | `llms/anthropic/experimental_pass_through/messages/transformation.py` |
|
||||||
|
| `/v1/messages` (passthrough) | Bedrock | `llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py` |
|
||||||
|
| `/v1/messages` (passthrough) | Vertex AI | `llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py` |
|
||||||
|
| Passthrough endpoints | All | `proxy/pass_through_endpoints/llm_provider_handlers/` |
|
||||||
|
|
||||||
|
### Example: Debugging prompt caching
|
||||||
|
|
||||||
|
If `/v1/messages` → Bedrock Converse prompt caching isn't working but Bedrock Invoke works:
|
||||||
|
|
||||||
|
1. **Bedrock Converse translation**: `llms/bedrock/chat/converse_transformation.py`
|
||||||
|
2. **Bedrock Invoke translation**: `llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py`
|
||||||
|
3. Compare how each handles `cache_control` in `transform_request()`
|
||||||
|
|
||||||
|
### How translations work
|
||||||
|
|
||||||
|
Each provider has a `Config` class that inherits from `BaseConfig` (`llms/base_llm/chat/transformation.py`):
|
||||||
|
|
||||||
|
```python
|
||||||
|
class ProviderConfig(BaseConfig):
|
||||||
|
def transform_request(self, model, messages, optional_params, litellm_params, headers):
|
||||||
|
# Convert OpenAI format → Provider format
|
||||||
|
return {"messages": transformed_messages, ...}
|
||||||
|
|
||||||
|
def transform_response(self, model, raw_response, model_response, logging_obj, ...):
|
||||||
|
# Convert Provider format → OpenAI format
|
||||||
|
return ModelResponse(choices=[...], usage=Usage(...))
|
||||||
|
```
|
||||||
|
|
||||||
|
The `BaseLLMHTTPHandler` (`llms/custom_httpx/llm_http_handler.py`) calls these methods - you never need to modify the handler itself.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Adding/Modifying Providers
|
||||||
|
|
||||||
|
### To add a new provider:
|
||||||
|
|
||||||
|
1. Create `llms/{provider}/chat/transformation.py`
|
||||||
|
2. Implement `Config` class with `transform_request()` and `transform_response()`
|
||||||
|
3. Add tests in `tests/llm_translation/test_{provider}.py`
|
||||||
|
|
||||||
|
### To add a feature (e.g., prompt caching):
|
||||||
|
|
||||||
|
1. Find the translation file from the table above
|
||||||
|
2. Modify `transform_request()` to handle the new parameter
|
||||||
|
3. Add unit tests that verify the transformation
|
||||||
|
|
||||||
|
### Testing checklist
|
||||||
|
|
||||||
|
When adding a feature, verify it works across all paths:
|
||||||
|
|
||||||
|
| Test | File Pattern |
|
||||||
|
|------|--------------|
|
||||||
|
| OpenAI passthrough | `tests/llm_translation/test_openai*.py` |
|
||||||
|
| Anthropic direct | `tests/llm_translation/test_anthropic*.py` |
|
||||||
|
| Bedrock Invoke | `tests/llm_translation/test_bedrock*.py` |
|
||||||
|
| Bedrock Converse | `tests/llm_translation/test_bedrock*converse*.py` |
|
||||||
|
| Vertex AI | `tests/llm_translation/test_vertex*.py` |
|
||||||
|
| Gemini | `tests/llm_translation/test_gemini*.py` |
|
||||||
|
|
||||||
|
### Unit testing translations
|
||||||
|
|
||||||
|
Translations are designed to be unit testable without making API calls:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm.llms.bedrock.chat.converse_transformation import BedrockConverseConfig
|
||||||
|
|
||||||
|
def test_prompt_caching_transform():
|
||||||
|
config = BedrockConverseConfig()
|
||||||
|
result = config.transform_request(
|
||||||
|
model="anthropic.claude-3-opus",
|
||||||
|
messages=[{"role": "user", "content": "test", "cache_control": {"type": "ephemeral"}}],
|
||||||
|
optional_params={},
|
||||||
|
litellm_params={},
|
||||||
|
headers={}
|
||||||
|
)
|
||||||
|
assert "cachePoint" in str(result) # Verify cache_control was translated
|
||||||
|
```
|
||||||
|
|
@ -90,6 +90,7 @@ LiteLLM is a unified interface for 100+ LLM providers with two main components:
|
||||||
- Pydantic v2 for data validation
|
- Pydantic v2 for data validation
|
||||||
- Async/await patterns throughout
|
- Async/await patterns throughout
|
||||||
- Type hints required for all public APIs
|
- Type hints required for all public APIs
|
||||||
|
- **Avoid imports within methods** — place all imports at the top of the file (module-level). Inline imports inside functions/methods make dependencies harder to trace and hurt readability. The only exception is avoiding circular imports where absolutely necessary.
|
||||||
|
|
||||||
### Testing Strategy
|
### Testing Strategy
|
||||||
- Unit tests in `tests/test_litellm/`
|
- Unit tests in `tests/test_litellm/`
|
||||||
|
|
|
||||||
|
|
@ -7,11 +7,20 @@ Thank you for your interest in contributing to LiteLLM! We welcome contributions
|
||||||
Here are the core requirements for any PR submitted to LiteLLM:
|
Here are the core requirements for any PR submitted to LiteLLM:
|
||||||
|
|
||||||
- [ ] **Sign the Contributor License Agreement (CLA)** - [see details](#contributor-license-agreement-cla)
|
- [ ] **Sign the Contributor License Agreement (CLA)** - [see details](#contributor-license-agreement-cla)
|
||||||
|
- [ ] **Keep scope isolated** - Your changes should address 1 specific problem at a time
|
||||||
|
|
||||||
|
#### Proxy (Backend) PRs
|
||||||
|
|
||||||
- [ ] **Add testing** - Adding at least 1 test is a hard requirement - [see details](#adding-testing)
|
- [ ] **Add testing** - Adding at least 1 test is a hard requirement - [see details](#adding-testing)
|
||||||
- [ ] **Ensure your PR passes all checks**:
|
- [ ] **Ensure your PR passes all checks**:
|
||||||
- [ ] [Unit Tests](#running-unit-tests) - `make test-unit`
|
- [ ] [Unit Tests](#running-unit-tests) - `make test-unit`
|
||||||
- [ ] [Linting / Formatting](#running-linting-and-formatting-checks) - `make lint`
|
- [ ] [Linting / Formatting](#running-linting-and-formatting-checks) - `make lint`
|
||||||
- [ ] **Keep scope isolated** - Your changes should address 1 specific problem at a time
|
|
||||||
|
#### UI PRs
|
||||||
|
|
||||||
|
- [ ] **Ensure the UI builds successfully** - `npm run build`
|
||||||
|
- [ ] **Ensure all UI unit tests pass** - `npm run test`
|
||||||
|
- [ ] **Add tests for new components or logic** - If you are adding a new component or new logic, add corresponding tests
|
||||||
|
|
||||||
## **Contributor License Agreement (CLA)**
|
## **Contributor License Agreement (CLA)**
|
||||||
|
|
||||||
|
|
@ -245,6 +254,43 @@ docker run \
|
||||||
--config /app/config.yaml --detailed_debug
|
--config /app/config.yaml --detailed_debug
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## UI Development
|
||||||
|
|
||||||
|
### 1. Setup Your Local UI Development Environment
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Clone the repo (if you haven't already)
|
||||||
|
git clone https://github.com/YOUR_USERNAME/litellm.git
|
||||||
|
cd litellm
|
||||||
|
|
||||||
|
# Navigate to the UI dashboard directory
|
||||||
|
cd ui/litellm-dashboard
|
||||||
|
|
||||||
|
# Install dependencies
|
||||||
|
npm install
|
||||||
|
|
||||||
|
# Start the development server
|
||||||
|
npm run dev
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Adding UI Tests
|
||||||
|
|
||||||
|
If you are adding a **new component** or **new logic**, you must add corresponding tests.
|
||||||
|
|
||||||
|
### 3. Running UI Unit Tests
|
||||||
|
|
||||||
|
```bash
|
||||||
|
npm run test
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4. Building the UI
|
||||||
|
|
||||||
|
Ensure the UI builds successfully before submitting your PR:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
npm run build
|
||||||
|
```
|
||||||
|
|
||||||
## Submitting Your PR
|
## Submitting Your PR
|
||||||
|
|
||||||
1. **Push your branch**: `git push origin your-feature-branch`
|
1. **Push your branch**: `git push origin your-feature-branch`
|
||||||
|
|
|
||||||
56
Dockerfile
56
Dockerfile
|
|
@ -3,6 +3,7 @@ ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||||
|
|
||||||
# Runtime image
|
# Runtime image
|
||||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
|
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||||
|
|
||||||
# Builder stage
|
# Builder stage
|
||||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||||
|
|
||||||
|
|
@ -20,7 +21,8 @@ RUN python -m pip install build
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# Build Admin UI
|
# Build Admin UI
|
||||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
# Convert Windows line endings to Unix and make executable
|
||||||
|
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||||
|
|
||||||
# Build the package
|
# Build the package
|
||||||
RUN rm -rf dist/* && python -m build
|
RUN rm -rf dist/* && python -m build
|
||||||
|
|
@ -45,8 +47,24 @@ FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||||
# Ensure runtime stage runs as root
|
# Ensure runtime stage runs as root
|
||||||
USER root
|
USER root
|
||||||
|
|
||||||
# Install runtime dependencies
|
# Install runtime dependencies (libsndfile needed for audio processing on ARM64)
|
||||||
RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip
|
RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip libsndfile && \
|
||||||
|
npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \
|
||||||
|
# SECURITY FIX: npm bundles tar, glob, and brace-expansion at multiple nested
|
||||||
|
# levels inside its dependency tree. `npm install -g <pkg>` only creates a
|
||||||
|
# SEPARATE global package, it does NOT replace npm's internal copies.
|
||||||
|
# We must find and replace EVERY copy inside npm's directory.
|
||||||
|
GLOBAL="$(npm root -g)" && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done && \
|
||||||
|
npm cache clean --force
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
# Copy the current directory contents into the container at /app
|
# Copy the current directory contents into the container at /app
|
||||||
|
|
@ -60,17 +78,37 @@ COPY --from=builder /wheels/ /wheels/
|
||||||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||||
|
|
||||||
|
# Replace the nodejs-wheel-binaries bundled node with the system node (fixes CVE-2025-55130)
|
||||||
|
RUN NODEJS_WHEEL_NODE=$(find /usr/lib -path "*/nodejs_wheel/bin/node" 2>/dev/null) && \
|
||||||
|
if [ -n "$NODEJS_WHEEL_NODE" ]; then cp /usr/bin/node "$NODEJS_WHEEL_NODE"; fi
|
||||||
|
|
||||||
# Remove test files and keys from dependencies
|
# Remove test files and keys from dependencies
|
||||||
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
||||||
find /usr/lib -type d -path "*/tornado/test" -delete
|
find /usr/lib -type d -path "*/tornado/test" -delete
|
||||||
|
|
||||||
# Install semantic_router and aurelio-sdk using script
|
# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete
|
||||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/.
|
||||||
|
# Patch every copy of tar, glob, and brace-expansion inside that tree.
|
||||||
|
RUN GLOBAL="$(npm root -g)" && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done
|
||||||
|
|
||||||
# Generate prisma client
|
# Install semantic_router and aurelio-sdk using script
|
||||||
RUN prisma generate
|
# Convert Windows line endings to Unix and make executable
|
||||||
RUN chmod +x docker/entrypoint.sh
|
RUN sed -i 's/\r$//' docker/install_auto_router.sh && chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
||||||
RUN chmod +x docker/prod_entrypoint.sh
|
|
||||||
|
# Generate prisma client using the correct schema
|
||||||
|
RUN prisma generate --schema=./litellm/proxy/schema.prisma
|
||||||
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
|
||||||
|
RUN sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||||
|
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
||||||
|
|
|
||||||
110
Makefile
110
Makefile
|
|
@ -1,7 +1,12 @@
|
||||||
# LiteLLM Makefile
|
# LiteLLM Makefile
|
||||||
# Simple Makefile for running tests and basic development tasks
|
# Simple Makefile for running tests and basic development tasks
|
||||||
|
|
||||||
.PHONY: help test test-unit test-integration test-unit-helm lint format install-dev install-proxy-dev install-test-deps install-helm-unittest check-circular-imports check-import-safety
|
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc \
|
||||||
|
test-unit-integrations test-unit-core-utils test-unit-other test-unit-root \
|
||||||
|
test-proxy-unit-a test-proxy-unit-b test-integration test-unit-helm \
|
||||||
|
info lint lint-dev format \
|
||||||
|
install-dev install-proxy-dev install-test-deps \
|
||||||
|
install-helm-unittest check-circular-imports check-import-safety
|
||||||
|
|
||||||
# Default target
|
# Default target
|
||||||
help:
|
help:
|
||||||
|
|
@ -22,9 +27,26 @@ help:
|
||||||
@echo " make check-import-safety - Check import safety"
|
@echo " make check-import-safety - Check import safety"
|
||||||
@echo " make test - Run all tests"
|
@echo " make test - Run all tests"
|
||||||
@echo " make test-unit - Run unit tests (tests/test_litellm)"
|
@echo " make test-unit - Run unit tests (tests/test_litellm)"
|
||||||
|
@echo " make test-unit-llms - Run LLM provider tests (~225 files)"
|
||||||
|
@echo " make test-unit-proxy-guardrails - Run proxy guardrails+mgmt tests (~51 files)"
|
||||||
|
@echo " make test-unit-proxy-core - Run proxy auth+client+db+hooks tests (~52 files)"
|
||||||
|
@echo " make test-unit-proxy-misc - Run proxy misc tests (~77 files)"
|
||||||
|
@echo " make test-unit-integrations - Run integration tests (~60 files)"
|
||||||
|
@echo " make test-unit-core-utils - Run core utils tests (~32 files)"
|
||||||
|
@echo " make test-unit-other - Run other tests (caching, responses, etc., ~69 files)"
|
||||||
|
@echo " make test-unit-root - Run root-level tests (~34 files)"
|
||||||
|
@echo " make test-proxy-unit-a - Run proxy_unit_tests (a-o, ~20 files)"
|
||||||
|
@echo " make test-proxy-unit-b - Run proxy_unit_tests (p-z, ~28 files)"
|
||||||
@echo " make test-integration - Run integration tests"
|
@echo " make test-integration - Run integration tests"
|
||||||
@echo " make test-unit-helm - Run helm unit tests"
|
@echo " make test-unit-helm - Run helm unit tests"
|
||||||
|
|
||||||
|
# Keep PIP simple for edge cases:
|
||||||
|
PIP := $(shell command -v pip > /dev/null 2>&1 && echo "pip" || echo "python3 -m pip")
|
||||||
|
|
||||||
|
# Show info
|
||||||
|
info:
|
||||||
|
@echo "PIP: $(PIP)"
|
||||||
|
|
||||||
# Installation targets
|
# Installation targets
|
||||||
install-dev:
|
install-dev:
|
||||||
poetry install --with dev
|
poetry install --with dev
|
||||||
|
|
@ -34,18 +56,19 @@ install-proxy-dev:
|
||||||
|
|
||||||
# CI-compatible installations (matches GitHub workflows exactly)
|
# CI-compatible installations (matches GitHub workflows exactly)
|
||||||
install-dev-ci:
|
install-dev-ci:
|
||||||
pip install openai==2.8.0
|
$(PIP) install openai==2.8.0
|
||||||
poetry install --with dev
|
poetry install --with dev
|
||||||
pip install openai==2.8.0
|
$(PIP) install openai==2.8.0
|
||||||
|
|
||||||
install-proxy-dev-ci:
|
install-proxy-dev-ci:
|
||||||
poetry install --with dev,proxy-dev --extras proxy
|
poetry install --with dev,proxy-dev --extras proxy
|
||||||
pip install openai==2.8.0
|
$(PIP) install openai==2.8.0
|
||||||
|
|
||||||
install-test-deps: install-proxy-dev
|
install-test-deps: install-proxy-dev
|
||||||
poetry run pip install "pytest-retry==1.6.3"
|
poetry run $(PIP) install "pytest-retry==1.6.3"
|
||||||
poetry run pip install pytest-xdist
|
poetry run $(PIP) install pytest-xdist
|
||||||
cd enterprise && poetry run pip install -e . && cd ..
|
poetry run $(PIP) install openapi-core
|
||||||
|
cd enterprise && poetry run $(PIP) install -e . && cd ..
|
||||||
|
|
||||||
install-helm-unittest:
|
install-helm-unittest:
|
||||||
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4 || echo "ignore error if plugin exists"
|
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4 || echo "ignore error if plugin exists"
|
||||||
|
|
@ -61,8 +84,40 @@ format-check: install-dev
|
||||||
lint-ruff: install-dev
|
lint-ruff: install-dev
|
||||||
cd litellm && poetry run ruff check . && cd ..
|
cd litellm && poetry run ruff check . && cd ..
|
||||||
|
|
||||||
|
# faster linter for developing ...
|
||||||
|
# inspiration from:
|
||||||
|
# https://github.com/astral-sh/ruff/discussions/10977
|
||||||
|
# https://github.com/astral-sh/ruff/discussions/4049
|
||||||
|
lint-format-changed: install-dev
|
||||||
|
@git diff origin/main --unified=0 --no-color -- '*.py' | \
|
||||||
|
perl -ne '\
|
||||||
|
if (/^diff --git a\/(.*) b\//) { $$file = $$1; } \
|
||||||
|
if (/^@@ .* \+(\d+)(?:,(\d+))? @@/) { \
|
||||||
|
$$start = $$1; $$count = $$2 || 1; $$end = $$start + $$count - 1; \
|
||||||
|
print "$$file:$$start:1-$$end:999\n"; \
|
||||||
|
}' | \
|
||||||
|
while read range; do \
|
||||||
|
file="$${range%%:*}"; \
|
||||||
|
lines="$${range#*:}"; \
|
||||||
|
echo "Formatting $$file (lines $$lines)"; \
|
||||||
|
poetry run ruff format --range "$$lines" "$$file"; \
|
||||||
|
done
|
||||||
|
|
||||||
|
lint-ruff-dev: install-dev
|
||||||
|
@tmpfile=$$(mktemp /tmp/ruff-dev.XXXXXX) && \
|
||||||
|
cd litellm && \
|
||||||
|
(poetry run ruff check . --output-format=pylint || true) > "$$tmpfile" && \
|
||||||
|
poetry run diff-quality --violations=pylint "$$tmpfile" --compare-branch=origin/main && \
|
||||||
|
cd .. ; \
|
||||||
|
rm -f "$$tmpfile"
|
||||||
|
|
||||||
|
lint-ruff-FULL-dev: install-dev
|
||||||
|
@files=$$(git diff --name-only origin/main -- '*.py'); \
|
||||||
|
if [ -n "$$files" ]; then echo "$$files" | xargs poetry run ruff check; \
|
||||||
|
else echo "No changed .py files to check."; fi
|
||||||
|
|
||||||
lint-mypy: install-dev
|
lint-mypy: install-dev
|
||||||
poetry run pip install types-requests types-setuptools types-redis types-PyYAML
|
poetry run $(PIP) install types-requests types-setuptools types-redis types-PyYAML
|
||||||
cd litellm && poetry run mypy . --ignore-missing-imports && cd ..
|
cd litellm && poetry run mypy . --ignore-missing-imports && cd ..
|
||||||
|
|
||||||
lint-black: format-check
|
lint-black: format-check
|
||||||
|
|
@ -71,11 +126,14 @@ check-circular-imports: install-dev
|
||||||
cd litellm && poetry run python ../tests/documentation_tests/test_circular_imports.py && cd ..
|
cd litellm && poetry run python ../tests/documentation_tests/test_circular_imports.py && cd ..
|
||||||
|
|
||||||
check-import-safety: install-dev
|
check-import-safety: install-dev
|
||||||
poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
|
@poetry run python -c "from litellm import *; print('[from litellm import *] OK! no issues!');" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
|
||||||
|
|
||||||
# Combined linting (matches test-linting.yml workflow)
|
# Combined linting (matches test-linting.yml workflow)
|
||||||
lint: format-check lint-ruff lint-mypy check-circular-imports check-import-safety
|
lint: format-check lint-ruff lint-mypy check-circular-imports check-import-safety
|
||||||
|
|
||||||
|
# Faster linting for local development (only checks changed code)
|
||||||
|
lint-dev: lint-format-changed lint-mypy check-circular-imports check-import-safety
|
||||||
|
|
||||||
# Testing targets
|
# Testing targets
|
||||||
test:
|
test:
|
||||||
poetry run pytest tests/
|
poetry run pytest tests/
|
||||||
|
|
@ -83,6 +141,38 @@ test:
|
||||||
test-unit: install-test-deps
|
test-unit: install-test-deps
|
||||||
poetry run pytest tests/test_litellm -x -vv -n 4
|
poetry run pytest tests/test_litellm -x -vv -n 4
|
||||||
|
|
||||||
|
# Matrix test targets (matching CI workflow groups)
|
||||||
|
test-unit-llms: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/llms --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-proxy-guardrails: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/proxy/guardrails tests/test_litellm/proxy/management_endpoints tests/test_litellm/proxy/management_helpers --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-proxy-core: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/proxy/auth tests/test_litellm/proxy/client tests/test_litellm/proxy/db tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-proxy-misc: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/proxy/_experimental tests/test_litellm/proxy/agent_endpoints tests/test_litellm/proxy/anthropic_endpoints tests/test_litellm/proxy/common_utils tests/test_litellm/proxy/discovery_endpoints tests/test_litellm/proxy/experimental tests/test_litellm/proxy/google_endpoints tests/test_litellm/proxy/health_endpoints tests/test_litellm/proxy/image_endpoints tests/test_litellm/proxy/middleware tests/test_litellm/proxy/openai_files_endpoint tests/test_litellm/proxy/pass_through_endpoints tests/test_litellm/proxy/prompts tests/test_litellm/proxy/public_endpoints tests/test_litellm/proxy/response_api_endpoints tests/test_litellm/proxy/spend_tracking tests/test_litellm/proxy/ui_crud_endpoints tests/test_litellm/proxy/vector_store_endpoints tests/test_litellm/proxy/test_*.py --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-integrations: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/integrations --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-core-utils: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/litellm_core_utils --tb=short -vv -n 2 --durations=20
|
||||||
|
|
||||||
|
test-unit-other: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/caching tests/test_litellm/responses tests/test_litellm/secret_managers tests/test_litellm/vector_stores tests/test_litellm/a2a_protocol tests/test_litellm/anthropic_interface tests/test_litellm/completion_extras tests/test_litellm/containers tests/test_litellm/enterprise tests/test_litellm/experimental_mcp_client tests/test_litellm/google_genai tests/test_litellm/images tests/test_litellm/interactions tests/test_litellm/passthrough tests/test_litellm/router_strategy tests/test_litellm/router_utils tests/test_litellm/types --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
test-unit-root: install-test-deps
|
||||||
|
poetry run pytest tests/test_litellm/test_*.py --tb=short -vv -n 4 --durations=20
|
||||||
|
|
||||||
|
# Proxy unit tests (tests/proxy_unit_tests split alphabetically)
|
||||||
|
test-proxy-unit-a: install-test-deps
|
||||||
|
poetry run pytest tests/proxy_unit_tests/test_[a-o]*.py --tb=short -vv -n 2 --durations=20
|
||||||
|
|
||||||
|
test-proxy-unit-b: install-test-deps
|
||||||
|
poetry run pytest tests/proxy_unit_tests/test_[p-z]*.py --tb=short -vv -n 2 --durations=20
|
||||||
|
|
||||||
test-integration:
|
test-integration:
|
||||||
poetry run pytest tests/ -k "not test_litellm"
|
poetry run pytest tests/ -k "not test_litellm"
|
||||||
|
|
||||||
|
|
@ -100,4 +190,4 @@ test-llm-translation-single: install-test-deps
|
||||||
@mkdir -p test-results
|
@mkdir -p test-results
|
||||||
poetry run pytest tests/llm_translation/$(FILE) \
|
poetry run pytest tests/llm_translation/$(FILE) \
|
||||||
--junitxml=test-results/junit.xml \
|
--junitxml=test-results/junit.xml \
|
||||||
-v --tb=short --maxfail=100 --timeout=300
|
-v --tb=short --maxfail=100 --timeout=300
|
||||||
|
|
|
||||||
449
README.md
449
README.md
|
|
@ -2,16 +2,16 @@
|
||||||
🚅 LiteLLM
|
🚅 LiteLLM
|
||||||
</h1>
|
</h1>
|
||||||
<p align="center">
|
<p align="center">
|
||||||
|
<p align="center">Call 100+ LLMs in OpenAI format. [Bedrock, Azure, OpenAI, VertexAI, Anthropic, Groq, etc.]
|
||||||
|
</p>
|
||||||
<p align="center">
|
<p align="center">
|
||||||
<a href="https://render.com/deploy?repo=https://github.com/BerriAI/litellm" target="_blank" rel="nofollow"><img src="https://render.com/images/deploy-to-render-button.svg" alt="Deploy to Render"></a>
|
<a href="https://render.com/deploy?repo=https://github.com/BerriAI/litellm" target="_blank" rel="nofollow"><img src="https://render.com/images/deploy-to-render-button.svg" alt="Deploy to Render"></a>
|
||||||
<a href="https://railway.app/template/HLP0Ub?referralCode=jch2ME">
|
<a href="https://railway.app/template/HLP0Ub?referralCode=jch2ME">
|
||||||
<img src="https://railway.app/button.svg" alt="Deploy on Railway">
|
<img src="https://railway.app/button.svg" alt="Deploy on Railway">
|
||||||
</a>
|
</a>
|
||||||
</p>
|
</p>
|
||||||
<p align="center">Call all LLM APIs using the OpenAI format [Bedrock, Huggingface, VertexAI, TogetherAI, Azure, OpenAI, Groq etc.]
|
|
||||||
<br>
|
|
||||||
</p>
|
</p>
|
||||||
<h4 align="center"><a href="https://docs.litellm.ai/docs/simple_proxy" target="_blank">LiteLLM Proxy Server (LLM Gateway)</a> | <a href="https://docs.litellm.ai/docs/enterprise#hosted-litellm-proxy" target="_blank"> Hosted Proxy</a> | <a href="https://docs.litellm.ai/docs/enterprise"target="_blank">Enterprise Tier</a></h4>
|
<h4 align="center"><a href="https://docs.litellm.ai/docs/simple_proxy" target="_blank">LiteLLM Proxy Server (AI Gateway)</a> | <a href="https://docs.litellm.ai/docs/enterprise#hosted-litellm-proxy" target="_blank"> Hosted Proxy</a> | <a href="https://docs.litellm.ai/docs/enterprise"target="_blank">Enterprise Tier</a></h4>
|
||||||
<h4 align="center">
|
<h4 align="center">
|
||||||
<a href="https://pypi.org/project/litellm/" target="_blank">
|
<a href="https://pypi.org/project/litellm/" target="_blank">
|
||||||
<img src="https://img.shields.io/pypi/v/litellm.svg" alt="PyPI Version">
|
<img src="https://img.shields.io/pypi/v/litellm.svg" alt="PyPI Version">
|
||||||
|
|
@ -30,27 +30,17 @@
|
||||||
</a>
|
</a>
|
||||||
</h4>
|
</h4>
|
||||||
|
|
||||||
LiteLLM manages:
|
<img width="2688" height="1600" alt="Group 7154 (1)" src="https://github.com/user-attachments/assets/c5ee0412-6fb5-4fb6-ab5b-bafae4209ca6" />
|
||||||
|
|
||||||
- Translate inputs to provider's `completion`, `embedding`, and `image_generation` endpoints
|
|
||||||
- [Consistent output](https://docs.litellm.ai/docs/completion/output), text responses will always be available at `['choices'][0]['message']['content']`
|
|
||||||
- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing)
|
|
||||||
- Set Budgets & Rate limits per project, api key, model [LiteLLM Proxy Server (LLM Gateway)](https://docs.litellm.ai/docs/simple_proxy)
|
|
||||||
|
|
||||||
LiteLLM Performance: **8ms P95 latency** at 1k RPS (See benchmarks [here](https://docs.litellm.ai/docs/benchmarks))
|
## Use LiteLLM for
|
||||||
|
|
||||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#litellm-proxy-server-llm-gateway---docs) <br>
|
<details open>
|
||||||
[**Jump to Supported LLM Providers**](https://docs.litellm.ai/docs/providers)
|
<summary><b>LLMs</b> - Call 100+ LLMs (Python SDK + AI Gateway)</summary>
|
||||||
|
|
||||||
🚨 **Stable Release:** Use docker images with the `-stable` tag. These have undergone 12 hour load tests, before being published. [More information about the release cycle here](https://docs.litellm.ai/docs/proxy/release_cycle)
|
[**All Supported Endpoints**](https://docs.litellm.ai/docs/supported_endpoints) - `/chat/completions`, `/responses`, `/embeddings`, `/images`, `/audio`, `/batches`, `/rerank`, `/a2a`, `/messages` and more.
|
||||||
|
|
||||||
Support for more providers. Missing a provider or LLM Platform, raise a [feature request](https://github.com/BerriAI/litellm/issues/new?assignees=&labels=enhancement&projects=&template=feature_request.yml&title=%5BFeature%5D%3A+).
|
### Python SDK
|
||||||
|
|
||||||
# Usage ([**Docs**](https://docs.litellm.ai/docs/))
|
|
||||||
|
|
||||||
<a target="_blank" href="https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/liteLLM_Getting_Started.ipynb">
|
|
||||||
<img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab"/>
|
|
||||||
</a>
|
|
||||||
|
|
||||||
```shell
|
```shell
|
||||||
pip install litellm
|
pip install litellm
|
||||||
|
|
@ -60,257 +50,237 @@ pip install litellm
|
||||||
from litellm import completion
|
from litellm import completion
|
||||||
import os
|
import os
|
||||||
|
|
||||||
## set ENV variables
|
|
||||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-key"
|
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-key"
|
||||||
|
|
||||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
# OpenAI
|
||||||
|
response = completion(model="openai/gpt-4o", messages=[{"role": "user", "content": "Hello!"}])
|
||||||
|
|
||||||
# openai call
|
# Anthropic
|
||||||
response = completion(model="openai/gpt-4o", messages=messages)
|
response = completion(model="anthropic/claude-sonnet-4-20250514", messages=[{"role": "user", "content": "Hello!"}])
|
||||||
|
|
||||||
# anthropic call
|
|
||||||
response = completion(model="anthropic/claude-sonnet-4-20250514", messages=messages)
|
|
||||||
print(response)
|
|
||||||
```
|
```
|
||||||
|
|
||||||
### Response (OpenAI Format)
|
### AI Gateway (Proxy Server)
|
||||||
|
|
||||||
```json
|
[**Getting Started - E2E Tutorial**](https://docs.litellm.ai/docs/proxy/docker_quick_start) - Setup virtual keys, make your first request
|
||||||
{
|
|
||||||
"id": "chatcmpl-1214900a-6cdd-4148-b663-b5e2f642b4de",
|
|
||||||
"created": 1751494488,
|
|
||||||
"model": "claude-sonnet-4-20250514",
|
|
||||||
"object": "chat.completion",
|
|
||||||
"system_fingerprint": null,
|
|
||||||
"choices": [
|
|
||||||
{
|
|
||||||
"finish_reason": "stop",
|
|
||||||
"index": 0,
|
|
||||||
"message": {
|
|
||||||
"content": "Hello! I'm doing well, thank you for asking. I'm here and ready to help with whatever you'd like to discuss or work on. How are you doing today?",
|
|
||||||
"role": "assistant",
|
|
||||||
"tool_calls": null,
|
|
||||||
"function_call": null
|
|
||||||
}
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"usage": {
|
|
||||||
"completion_tokens": 39,
|
|
||||||
"prompt_tokens": 13,
|
|
||||||
"total_tokens": 52,
|
|
||||||
"completion_tokens_details": null,
|
|
||||||
"prompt_tokens_details": {
|
|
||||||
"audio_tokens": null,
|
|
||||||
"cached_tokens": 0
|
|
||||||
},
|
|
||||||
"cache_creation_input_tokens": 0,
|
|
||||||
"cache_read_input_tokens": 0
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
> **Note:** LiteLLM also supports the [Responses API](https://docs.litellm.ai/docs/response_api) (`litellm.responses()`)
|
|
||||||
|
|
||||||
Call any model supported by a provider, with `model=<provider_name>/<model_name>`. There might be provider-specific details here, so refer to [provider docs for more information](https://docs.litellm.ai/docs/providers)
|
|
||||||
|
|
||||||
## Async ([Docs](https://docs.litellm.ai/docs/completion/stream#async-completion))
|
|
||||||
|
|
||||||
```python
|
|
||||||
from litellm import acompletion
|
|
||||||
import asyncio
|
|
||||||
|
|
||||||
async def test_get_response():
|
|
||||||
user_message = "Hello, how are you?"
|
|
||||||
messages = [{"content": user_message, "role": "user"}]
|
|
||||||
response = await acompletion(model="openai/gpt-4o", messages=messages)
|
|
||||||
return response
|
|
||||||
|
|
||||||
response = asyncio.run(test_get_response())
|
|
||||||
print(response)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Streaming ([Docs](https://docs.litellm.ai/docs/completion/stream))
|
|
||||||
|
|
||||||
LiteLLM supports streaming the model response back, pass `stream=True` to get a streaming iterator in response.
|
|
||||||
Streaming is supported for all models (Bedrock, Huggingface, TogetherAI, Azure, OpenAI, etc.)
|
|
||||||
|
|
||||||
```python
|
|
||||||
from litellm import completion
|
|
||||||
|
|
||||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
|
||||||
|
|
||||||
# gpt-4o
|
|
||||||
response = completion(model="openai/gpt-4o", messages=messages, stream=True)
|
|
||||||
for part in response:
|
|
||||||
print(part.choices[0].delta.content or "")
|
|
||||||
|
|
||||||
# claude sonnet 4
|
|
||||||
response = completion('anthropic/claude-sonnet-4-20250514', messages, stream=True)
|
|
||||||
for part in response:
|
|
||||||
print(part)
|
|
||||||
```
|
|
||||||
|
|
||||||
### Response chunk (OpenAI Format)
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"id": "chatcmpl-fe575c37-5004-4926-ae5e-bfbc31f356ca",
|
|
||||||
"created": 1751494808,
|
|
||||||
"model": "claude-sonnet-4-20250514",
|
|
||||||
"object": "chat.completion.chunk",
|
|
||||||
"system_fingerprint": null,
|
|
||||||
"choices": [
|
|
||||||
{
|
|
||||||
"finish_reason": null,
|
|
||||||
"index": 0,
|
|
||||||
"delta": {
|
|
||||||
"provider_specific_fields": null,
|
|
||||||
"content": "Hello",
|
|
||||||
"role": "assistant",
|
|
||||||
"function_call": null,
|
|
||||||
"tool_calls": null,
|
|
||||||
"audio": null
|
|
||||||
},
|
|
||||||
"logprobs": null
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"provider_specific_fields": null,
|
|
||||||
"stream_options": null,
|
|
||||||
"citations": null
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Logging Observability ([Docs](https://docs.litellm.ai/docs/observability/callbacks))
|
|
||||||
|
|
||||||
LiteLLM exposes pre defined callbacks to send data to Lunary, MLflow, Langfuse, DynamoDB, s3 Buckets, Helicone, Promptlayer, Traceloop, Athina, Slack
|
|
||||||
|
|
||||||
```python
|
|
||||||
from litellm import completion
|
|
||||||
|
|
||||||
## set env variables for logging tools (when using MLflow, no API key set up is required)
|
|
||||||
os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key"
|
|
||||||
os.environ["HELICONE_API_KEY"] = "your-helicone-auth-key"
|
|
||||||
os.environ["LANGFUSE_PUBLIC_KEY"] = ""
|
|
||||||
os.environ["LANGFUSE_SECRET_KEY"] = ""
|
|
||||||
os.environ["ATHINA_API_KEY"] = "your-athina-api-key"
|
|
||||||
|
|
||||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
|
||||||
|
|
||||||
# set callbacks
|
|
||||||
litellm.success_callback = ["lunary", "mlflow", "langfuse", "athina", "helicone"] # log input/output to lunary, langfuse, supabase, athina, helicone etc
|
|
||||||
|
|
||||||
#openai call
|
|
||||||
response = completion(model="openai/gpt-4o", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}])
|
|
||||||
```
|
|
||||||
|
|
||||||
# LiteLLM Proxy Server (LLM Gateway) - ([Docs](https://docs.litellm.ai/docs/simple_proxy))
|
|
||||||
|
|
||||||
Track spend + Load Balance across multiple projects
|
|
||||||
|
|
||||||
[Hosted Proxy](https://docs.litellm.ai/docs/enterprise#hosted-litellm-proxy)
|
|
||||||
|
|
||||||
The proxy provides:
|
|
||||||
|
|
||||||
1. [Hooks for auth](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth)
|
|
||||||
2. [Hooks for logging](https://docs.litellm.ai/docs/proxy/logging#step-1---create-your-custom-litellm-callback-class)
|
|
||||||
3. [Cost tracking](https://docs.litellm.ai/docs/proxy/virtual_keys#tracking-spend)
|
|
||||||
4. [Rate Limiting](https://docs.litellm.ai/docs/proxy/users#set-rate-limits)
|
|
||||||
|
|
||||||
## 📖 Proxy Endpoints - [Swagger Docs](https://litellm-api.up.railway.app/)
|
|
||||||
|
|
||||||
|
|
||||||
## Quick Start Proxy - CLI
|
|
||||||
|
|
||||||
```shell
|
```shell
|
||||||
pip install 'litellm[proxy]'
|
pip install 'litellm[proxy]'
|
||||||
|
litellm --model gpt-4o
|
||||||
```
|
```
|
||||||
|
|
||||||
### Step 1: Start litellm proxy
|
|
||||||
|
|
||||||
```shell
|
|
||||||
$ litellm --model huggingface/bigcode/starcoder
|
|
||||||
|
|
||||||
#INFO: Proxy running on http://0.0.0.0:4000
|
|
||||||
```
|
|
||||||
|
|
||||||
### Step 2: Make ChatCompletions Request to Proxy
|
|
||||||
|
|
||||||
|
|
||||||
> [!IMPORTANT]
|
|
||||||
> 💡 [Use LiteLLM Proxy with Langchain (Python, JS), OpenAI SDK (Python, JS) Anthropic SDK, Mistral SDK, LlamaIndex, Instructor, Curl](https://docs.litellm.ai/docs/proxy/user_keys)
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
import openai # openai v1.0.0+
|
import openai
|
||||||
client = openai.OpenAI(api_key="anything",base_url="http://0.0.0.0:4000") # set proxy to base_url
|
|
||||||
# request sent to model set on litellm proxy, `litellm --model`
|
|
||||||
response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
|
|
||||||
{
|
|
||||||
"role": "user",
|
|
||||||
"content": "this is a test request, write a short poem"
|
|
||||||
}
|
|
||||||
])
|
|
||||||
|
|
||||||
print(response)
|
client = openai.OpenAI(api_key="anything", base_url="http://0.0.0.0:4000")
|
||||||
|
response = client.chat.completions.create(
|
||||||
|
model="gpt-4o",
|
||||||
|
messages=[{"role": "user", "content": "Hello!"}]
|
||||||
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
## Proxy Key Management ([Docs](https://docs.litellm.ai/docs/proxy/virtual_keys))
|
[**Docs: LLM Providers**](https://docs.litellm.ai/docs/providers)
|
||||||
|
|
||||||
Connect the proxy with a Postgres DB to create proxy keys
|
</details>
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary><b>Agents</b> - Invoke A2A Agents (Python SDK + AI Gateway)</summary>
|
||||||
|
|
||||||
|
[**Supported Providers**](https://docs.litellm.ai/docs/a2a#add-a2a-agents) - LangGraph, Vertex AI Agent Engine, Azure AI Foundry, Bedrock AgentCore, Pydantic AI
|
||||||
|
|
||||||
|
### Python SDK - A2A Protocol
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm.a2a_protocol import A2AClient
|
||||||
|
from a2a.types import SendMessageRequest, MessageSendParams
|
||||||
|
from uuid import uuid4
|
||||||
|
|
||||||
|
client = A2AClient(base_url="http://localhost:10001")
|
||||||
|
|
||||||
|
request = SendMessageRequest(
|
||||||
|
id=str(uuid4()),
|
||||||
|
params=MessageSendParams(
|
||||||
|
message={
|
||||||
|
"role": "user",
|
||||||
|
"parts": [{"kind": "text", "text": "Hello!"}],
|
||||||
|
"messageId": uuid4().hex,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
response = await client.send_message(request)
|
||||||
|
```
|
||||||
|
|
||||||
|
### AI Gateway (Proxy Server)
|
||||||
|
|
||||||
|
**Step 1.** [Add your Agent to the AI Gateway](https://docs.litellm.ai/docs/a2a#adding-your-agent)
|
||||||
|
|
||||||
|
**Step 2.** Call Agent via A2A SDK
|
||||||
|
|
||||||
|
```python
|
||||||
|
from a2a.client import A2ACardResolver, A2AClient
|
||||||
|
from a2a.types import MessageSendParams, SendMessageRequest
|
||||||
|
from uuid import uuid4
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
base_url = "http://localhost:4000/a2a/my-agent" # LiteLLM proxy + agent name
|
||||||
|
headers = {"Authorization": "Bearer sk-1234"} # LiteLLM Virtual Key
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(headers=headers) as httpx_client:
|
||||||
|
resolver = A2ACardResolver(httpx_client=httpx_client, base_url=base_url)
|
||||||
|
agent_card = await resolver.get_agent_card()
|
||||||
|
client = A2AClient(httpx_client=httpx_client, agent_card=agent_card)
|
||||||
|
|
||||||
|
request = SendMessageRequest(
|
||||||
|
id=str(uuid4()),
|
||||||
|
params=MessageSendParams(
|
||||||
|
message={
|
||||||
|
"role": "user",
|
||||||
|
"parts": [{"kind": "text", "text": "Hello!"}],
|
||||||
|
"messageId": uuid4().hex,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
response = await client.send_message(request)
|
||||||
|
```
|
||||||
|
|
||||||
|
[**Docs: A2A Agent Gateway**](https://docs.litellm.ai/docs/a2a)
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary><b>MCP Tools</b> - Connect MCP servers to any LLM (Python SDK + AI Gateway)</summary>
|
||||||
|
|
||||||
|
### Python SDK - MCP Bridge
|
||||||
|
|
||||||
|
```python
|
||||||
|
from mcp import ClientSession, StdioServerParameters
|
||||||
|
from mcp.client.stdio import stdio_client
|
||||||
|
from litellm import experimental_mcp_client
|
||||||
|
import litellm
|
||||||
|
|
||||||
|
server_params = StdioServerParameters(command="python", args=["mcp_server.py"])
|
||||||
|
|
||||||
|
async with stdio_client(server_params) as (read, write):
|
||||||
|
async with ClientSession(read, write) as session:
|
||||||
|
await session.initialize()
|
||||||
|
|
||||||
|
# Load MCP tools in OpenAI format
|
||||||
|
tools = await experimental_mcp_client.load_mcp_tools(session=session, format="openai")
|
||||||
|
|
||||||
|
# Use with any LiteLLM model
|
||||||
|
response = await litellm.acompletion(
|
||||||
|
model="gpt-4o",
|
||||||
|
messages=[{"role": "user", "content": "What's 3 + 5?"}],
|
||||||
|
tools=tools
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
### AI Gateway - MCP Gateway
|
||||||
|
|
||||||
|
**Step 1.** [Add your MCP Server to the AI Gateway](https://docs.litellm.ai/docs/mcp#adding-your-mcp)
|
||||||
|
|
||||||
|
**Step 2.** Call MCP tools via `/chat/completions`
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Get the code
|
curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||||
git clone https://github.com/BerriAI/litellm
|
-H 'Authorization: Bearer sk-1234' \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
# Go to folder
|
-d '{
|
||||||
cd litellm
|
"model": "gpt-4o",
|
||||||
|
"messages": [{"role": "user", "content": "Summarize the latest open PR"}],
|
||||||
# Add the master key - you can change this after setup
|
"tools": [{
|
||||||
echo 'LITELLM_MASTER_KEY="sk-1234"' > .env
|
"type": "mcp",
|
||||||
|
"server_url": "litellm_proxy/mcp/github",
|
||||||
# Add the litellm salt key - you cannot change this after adding a model
|
"server_label": "github_mcp",
|
||||||
# It is used to encrypt / decrypt your LLM API Key credentials
|
"require_approval": "never"
|
||||||
# We recommend - https://1password.com/password-generator/
|
}]
|
||||||
# password generator to get a random hash for litellm salt key
|
}'
|
||||||
echo 'LITELLM_SALT_KEY="sk-1234"' >> .env
|
|
||||||
|
|
||||||
# Start
|
|
||||||
docker compose up
|
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Use with Cursor IDE
|
||||||
|
|
||||||
UI on `/ui` on your proxy server
|
```json
|
||||||

|
|
||||||
|
|
||||||
Set budgets and rate limits across multiple projects
|
|
||||||
`POST /key/generate`
|
|
||||||
|
|
||||||
### Request
|
|
||||||
|
|
||||||
```shell
|
|
||||||
curl 'http://0.0.0.0:4000/key/generate' \
|
|
||||||
--header 'Authorization: Bearer sk-1234' \
|
|
||||||
--header 'Content-Type: application/json' \
|
|
||||||
--data-raw '{"models": ["gpt-3.5-turbo", "gpt-4", "claude-2"], "duration": "20m","metadata": {"user": "ishaan@berri.ai", "team": "core-infra"}}'
|
|
||||||
```
|
|
||||||
|
|
||||||
### Expected Response
|
|
||||||
|
|
||||||
```shell
|
|
||||||
{
|
{
|
||||||
"key": "sk-kdEXbIqZRwEeEiHwdg7sFA", # Bearer token
|
"mcpServers": {
|
||||||
"expires": "2023-11-19T01:38:25.838000+00:00" # datetime object
|
"LiteLLM": {
|
||||||
|
"url": "http://localhost:4000/mcp",
|
||||||
|
"headers": {
|
||||||
|
"x-litellm-api-key": "Bearer sk-1234"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
[**Docs: MCP Gateway**](https://docs.litellm.ai/docs/mcp)
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How to use LiteLLM
|
||||||
|
|
||||||
|
You can use LiteLLM through either the Proxy Server or Python SDK. Both gives you a unified interface to access multiple LLMs (100+ LLMs). Choose the option that best fits your needs:
|
||||||
|
|
||||||
|
<table style={{width: '100%', tableLayout: 'fixed'}}>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th style={{width: '14%'}}></th>
|
||||||
|
<th style={{width: '43%'}}><strong><a href="https://docs.litellm.ai/docs/simple_proxy">LiteLLM AI Gateway</a></strong></th>
|
||||||
|
<th style={{width: '43%'}}><strong><a href="https://docs.litellm.ai/docs/">LiteLLM Python SDK</a></strong></th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
<tr>
|
||||||
|
<td style={{width: '14%'}}><strong>Use Case</strong></td>
|
||||||
|
<td style={{width: '43%'}}>Central service (LLM Gateway) to access multiple LLMs</td>
|
||||||
|
<td style={{width: '43%'}}>Use LiteLLM directly in your Python code</td>
|
||||||
|
</tr>
|
||||||
|
<tr>
|
||||||
|
<td style={{width: '14%'}}><strong>Who Uses It?</strong></td>
|
||||||
|
<td style={{width: '43%'}}>Gen AI Enablement / ML Platform Teams</td>
|
||||||
|
<td style={{width: '43%'}}>Developers building LLM projects</td>
|
||||||
|
</tr>
|
||||||
|
<tr>
|
||||||
|
<td style={{width: '14%'}}><strong>Key Features</strong></td>
|
||||||
|
<td style={{width: '43%'}}>Centralized API gateway with authentication and authorization, multi-tenant cost tracking and spend management per project/user, per-project customization (logging, guardrails, caching), virtual keys for secure access control, admin dashboard UI for monitoring and management</td>
|
||||||
|
<td style={{width: '43%'}}>Direct Python library integration in your codebase, Router with retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - <a href="https://docs.litellm.ai/docs/routing">Router</a>, application-level load balancing and cost tracking, exception handling with OpenAI-compatible errors, observability callbacks (Lunary, MLflow, Langfuse, etc.)</td>
|
||||||
|
</tr>
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
|
||||||
|
LiteLLM Performance: **8ms P95 latency** at 1k RPS (See benchmarks [here](https://docs.litellm.ai/docs/benchmarks))
|
||||||
|
|
||||||
|
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://docs.litellm.ai/docs/simple_proxy) <br>
|
||||||
|
[**Jump to Supported LLM Providers**](https://docs.litellm.ai/docs/providers)
|
||||||
|
|
||||||
|
**Stable Release:** Use docker images with the `-stable` tag. These have undergone 12 hour load tests, before being published. [More information about the release cycle here](https://docs.litellm.ai/docs/proxy/release_cycle)
|
||||||
|
|
||||||
|
Support for more providers. Missing a provider or LLM Platform, raise a [feature request](https://github.com/BerriAI/litellm/issues/new?assignees=&labels=enhancement&projects=&template=feature_request.yml&title=%5BFeature%5D%3A+).
|
||||||
|
|
||||||
|
## OSS Adopters
|
||||||
|
|
||||||
|
<table>
|
||||||
|
<tr>
|
||||||
|
<td><img height="60" alt="Stripe" src="https://github.com/user-attachments/assets/f7296d4f-9fbd-460d-9d05-e4df31697c4b" /></td>
|
||||||
|
<td><img height="60" alt="Google ADK" src="https://github.com/user-attachments/assets/caf270a2-5aee-45c4-8222-41a2070c4f19" /></td>
|
||||||
|
<td><img height="60" alt="Greptile" src="https://github.com/user-attachments/assets/0be4bd8a-7cfa-48d3-9090-f415fe948280" /></td>
|
||||||
|
<td><img height="60" alt="OpenHands" src="https://github.com/user-attachments/assets/a6150c4c-149e-4cae-888b-8b92be6e003f" /></td>
|
||||||
|
<td><h2>Netflix</h2></td>
|
||||||
|
<td><img height="60" alt="OpenAI Agents SDK" src="https://github.com/user-attachments/assets/c02f7be0-8c2e-4d27-aea7-7c024bfaebc0" /></td>
|
||||||
|
</tr>
|
||||||
|
</table>
|
||||||
|
|
||||||
## Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
|
## Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
|
||||||
|
|
||||||
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |
|
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |
|
||||||
|-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------|
|
|-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------|
|
||||||
|
| [Abliteration (`abliteration`)](https://docs.litellm.ai/docs/providers/abliteration) | ✅ | | | | | | | | | |
|
||||||
| [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
| [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||||
| [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
| [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
| [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [Aleph Alpha](https://docs.litellm.ai/docs/providers/aleph_alpha) | ✅ | ✅ | ✅ | | | | | | | |
|
| [Aleph Alpha](https://docs.litellm.ai/docs/providers/aleph_alpha) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
|
| [Amazon Nova](https://docs.litellm.ai/docs/providers/amazon_nova) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
| [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
||||||
| [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
| [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
||||||
| [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | |
|
| [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
|
|
@ -339,7 +309,7 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
||||||
| [Deepgram (`deepgram`)](https://docs.litellm.ai/docs/providers/deepgram) | ✅ | ✅ | ✅ | | | ✅ | | | | |
|
| [Deepgram (`deepgram`)](https://docs.litellm.ai/docs/providers/deepgram) | ✅ | ✅ | ✅ | | | ✅ | | | | |
|
||||||
| [DeepInfra (`deepinfra`)](https://docs.litellm.ai/docs/providers/deepinfra) | ✅ | ✅ | ✅ | | | | | | | |
|
| [DeepInfra (`deepinfra`)](https://docs.litellm.ai/docs/providers/deepinfra) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [Deepseek (`deepseek`)](https://docs.litellm.ai/docs/providers/deepseek) | ✅ | ✅ | ✅ | | | | | | | |
|
| [Deepseek (`deepseek`)](https://docs.litellm.ai/docs/providers/deepseek) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [ElevenLabs (`elevenlabs`)](https://docs.litellm.ai/docs/providers/elevenlabs) | ✅ | ✅ | ✅ | | | | ✅ | | | |
|
| [ElevenLabs (`elevenlabs`)](https://docs.litellm.ai/docs/providers/elevenlabs) | ✅ | ✅ | ✅ | | | ✅ | ✅ | | | |
|
||||||
| [Empower (`empower`)](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | | | | | | | |
|
| [Empower (`empower`)](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
| [Fal AI (`fal_ai`)](https://docs.litellm.ai/docs/providers/fal_ai) | ✅ | ✅ | ✅ | | ✅ | | | | | |
|
| [Fal AI (`fal_ai`)](https://docs.litellm.ai/docs/providers/fal_ai) | ✅ | ✅ | ✅ | | ✅ | | | | | |
|
||||||
| [Featherless AI (`featherless_ai`)](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
| [Featherless AI (`featherless_ai`)](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||||
|
|
@ -417,7 +387,9 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
||||||
1. (In root) create virtual environment `python -m venv .venv`
|
1. (In root) create virtual environment `python -m venv .venv`
|
||||||
2. Activate virtual environment `source .venv/bin/activate`
|
2. Activate virtual environment `source .venv/bin/activate`
|
||||||
3. Install dependencies `pip install -e ".[all]"`
|
3. Install dependencies `pip install -e ".[all]"`
|
||||||
4. Start proxy backend `python litellm/proxy_cli.py`
|
4. `pip install prisma`
|
||||||
|
5. `prisma generate`
|
||||||
|
6. Start proxy backend `python litellm/proxy/proxy_cli.py`
|
||||||
|
|
||||||
### Frontend
|
### Frontend
|
||||||
1. Navigate to `ui/litellm-dashboard`
|
1. Navigate to `ui/litellm-dashboard`
|
||||||
|
|
@ -499,4 +471,3 @@ All these checks must pass before your PR can be merged.
|
||||||
<img src="https://contrib.rocks/image?repo=BerriAI/litellm" />
|
<img src="https://contrib.rocks/image?repo=BerriAI/litellm" />
|
||||||
</a>
|
</a>
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,4 +0,0 @@
|
||||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello, how are you?"}]}}
|
|
||||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "What is the weather today?"}]}}
|
|
||||||
{"custom_id": "request-3", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Tell me a short joke"}]}}
|
|
||||||
|
|
||||||
36
ci_cd/.grype.yaml
Normal file
36
ci_cd/.grype.yaml
Normal file
|
|
@ -0,0 +1,36 @@
|
||||||
|
ignore:
|
||||||
|
- vulnerability: CVE-2026-22184
|
||||||
|
reason: no fixed zlib package is available yet in the Wolfi repositories, so this is ignored temporarily until an upstream release exists
|
||||||
|
# Wolfi base image: Python 3.13 and Node from apk have no fixed builds in Wolfi yet / not applicable
|
||||||
|
- vulnerability: CVE-2025-55130
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: CVE-2025-59465
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: CVE-2025-55131
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: CVE-2025-59466
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: CVE-2026-21637
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: CVE-2025-55132
|
||||||
|
reason: Node in Wolfi apk; only used for Admin UI build/prisma
|
||||||
|
- vulnerability: GHSA-hx9q-6w63-j58v
|
||||||
|
reason: orjson dumps recursion; allowlisted
|
||||||
|
- vulnerability: GHSA-73rr-hh4g-fpgx
|
||||||
|
reason: diff npm transitive dep; override in package.json, allowlisted
|
||||||
|
- vulnerability: CVE-2026-0865
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2025-15282
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2026-0672
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2025-15366
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2025-15367
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2025-11468
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2025-12781
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
|
- vulnerability: CVE-2026-1299
|
||||||
|
reason: Python 3.13 in Wolfi base; no fixed apk build yet
|
||||||
40
ci_cd/TEST_KEY_PATTERNS.md
Normal file
40
ci_cd/TEST_KEY_PATTERNS.md
Normal file
|
|
@ -0,0 +1,40 @@
|
||||||
|
# Test Key Patterns Standard
|
||||||
|
|
||||||
|
Standard patterns for test/mock keys and credentials in the LiteLLM codebase to avoid triggering secret detection.
|
||||||
|
|
||||||
|
## How GitGuardian Works
|
||||||
|
|
||||||
|
GitGuardian uses **machine learning and entropy analysis**, not just pattern matching:
|
||||||
|
- **Low entropy** values (like `sk-1234`, `postgres`) are automatically ignored
|
||||||
|
- **High entropy** values (realistic-looking secrets) trigger detection
|
||||||
|
- **Context-aware** detection understands code syntax like `os.environ["KEY"]`
|
||||||
|
|
||||||
|
## Recommended Test Key Patterns
|
||||||
|
|
||||||
|
### Option 1: Low Entropy Values (Simplest)
|
||||||
|
These won't trigger GitGuardian's ML detector:
|
||||||
|
|
||||||
|
```python
|
||||||
|
api_key = "sk-1234"
|
||||||
|
api_key = "sk-12345"
|
||||||
|
database_password = "postgres"
|
||||||
|
token = "test123"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Option 2: High Entropy with Test Prefixes
|
||||||
|
If you need realistic-looking test keys with high entropy, use these prefixes:
|
||||||
|
|
||||||
|
```python
|
||||||
|
api_key = "sk-test-abc123def456ghi789..." # OpenAI-style test key
|
||||||
|
api_key = "sk-mock-1234567890abcdef1234..." # Mock key
|
||||||
|
api_key = "sk-fake-xyz789uvw456rst123..." # Fake key
|
||||||
|
token = "test-api-key-with-high-entropy"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Configured Ignore Patterns
|
||||||
|
|
||||||
|
These patterns are in `.gitguardian.yaml` for high-entropy test keys:
|
||||||
|
- `sk-test-*` - OpenAI-style test keys
|
||||||
|
- `sk-mock-*` - Mock API keys
|
||||||
|
- `sk-fake-*` - Fake API keys
|
||||||
|
- `test-api-key` - Generic test tokens
|
||||||
|
|
@ -26,15 +26,65 @@ install_grype() {
|
||||||
echo "Grype installed successfully"
|
echo "Grype installed successfully"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Function to install ggshield
|
||||||
|
install_ggshield() {
|
||||||
|
echo "Installing ggshield..."
|
||||||
|
pip3 install --upgrade pip
|
||||||
|
pip3 install ggshield
|
||||||
|
echo "ggshield installed successfully"
|
||||||
|
}
|
||||||
|
|
||||||
|
# # Function to run secret detection scans
|
||||||
|
# run_secret_detection() {
|
||||||
|
# echo "Running secret detection scans..."
|
||||||
|
|
||||||
|
# if ! command -v ggshield &> /dev/null; then
|
||||||
|
# install_ggshield
|
||||||
|
# fi
|
||||||
|
|
||||||
|
# # Check if GITGUARDIAN_API_KEY is set (required for CI/CD)
|
||||||
|
# if [ -z "$GITGUARDIAN_API_KEY" ]; then
|
||||||
|
# echo "Warning: GITGUARDIAN_API_KEY environment variable is not set."
|
||||||
|
# echo "ggshield requires a GitGuardian API key to scan for secrets."
|
||||||
|
# echo "Please set GITGUARDIAN_API_KEY in your CI/CD environment variables."
|
||||||
|
# exit 1
|
||||||
|
# fi
|
||||||
|
|
||||||
|
# echo "Scanning codebase for secrets..."
|
||||||
|
# echo "Note: Large codebases may take several minutes due to API rate limits (50 requests/minute on free plan)"
|
||||||
|
# echo "ggshield will automatically handle rate limits and retry as needed."
|
||||||
|
# echo "Binary files, cache files, and build artifacts are excluded via .gitguardian.yaml"
|
||||||
|
|
||||||
|
# # Use --recursive for directory scanning and auto-confirm if prompted
|
||||||
|
# # .gitguardian.yaml will automatically exclude binary files, wheel files, etc.
|
||||||
|
# # GITGUARDIAN_API_KEY environment variable will be used for authentication
|
||||||
|
# echo y | ggshield secret scan path . --recursive || {
|
||||||
|
# echo ""
|
||||||
|
# echo "=========================================="
|
||||||
|
# echo "ERROR: Secret Detection Failed"
|
||||||
|
# echo "=========================================="
|
||||||
|
# echo "ggshield has detected secrets in the codebase."
|
||||||
|
# echo "Please review discovered secrets above, revoke any actively used secrets"
|
||||||
|
# echo "from underlying systems and make changes to inject secrets dynamically at runtime."
|
||||||
|
# echo ""
|
||||||
|
# echo "For more information, see: https://docs.gitguardian.com/secrets-detection/"
|
||||||
|
# echo "=========================================="
|
||||||
|
# echo ""
|
||||||
|
# exit 1
|
||||||
|
# }
|
||||||
|
|
||||||
|
# echo "Secret detection scans completed successfully"
|
||||||
|
# }
|
||||||
|
|
||||||
# Function to run Trivy scans
|
# Function to run Trivy scans
|
||||||
run_trivy_scans() {
|
run_trivy_scans() {
|
||||||
echo "Running Trivy scans..."
|
echo "Running Trivy scans..."
|
||||||
|
|
||||||
echo "Scanning LiteLLM Docs..."
|
echo "Scanning LiteLLM Docs..."
|
||||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
trivy fs --ignorefile .trivyignore --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||||
|
|
||||||
echo "Scanning LiteLLM UI..."
|
echo "Scanning LiteLLM UI..."
|
||||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
trivy fs --ignorefile .trivyignore --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||||
|
|
||||||
echo "Trivy scans completed successfully"
|
echo "Trivy scans completed successfully"
|
||||||
}
|
}
|
||||||
|
|
@ -51,12 +101,12 @@ run_grype_scans() {
|
||||||
# Build and scan Dockerfile.database
|
# Build and scan Dockerfile.database
|
||||||
echo "Building and scanning Dockerfile.database..."
|
echo "Building and scanning Dockerfile.database..."
|
||||||
docker build --no-cache -t litellm-database:latest -f ./docker/Dockerfile.database .
|
docker build --no-cache -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||||
grype litellm-database:latest --fail-on critical
|
grype litellm-database:latest --config ci_cd/.grype.yaml --fail-on critical
|
||||||
|
|
||||||
# Build and scan main Dockerfile
|
# Build and scan main Dockerfile
|
||||||
echo "Building and scanning main Dockerfile..."
|
echo "Building and scanning main Dockerfile..."
|
||||||
docker build --no-cache -t litellm:latest .
|
docker build --no-cache -t litellm:latest .
|
||||||
grype litellm:latest --fail-on critical
|
grype litellm:latest --config ci_cd/.grype.yaml --fail-on critical
|
||||||
|
|
||||||
# Restore original .dockerignore
|
# Restore original .dockerignore
|
||||||
echo "Restoring original .dockerignore..."
|
echo "Restoring original .dockerignore..."
|
||||||
|
|
@ -78,6 +128,36 @@ run_grype_scans() {
|
||||||
"GHSA-5j98-mcp5-4vw2"
|
"GHSA-5j98-mcp5-4vw2"
|
||||||
"CVE-2025-13836" # Python 3.13 HTTP response reading OOM/DoS - no fix available in base image
|
"CVE-2025-13836" # Python 3.13 HTTP response reading OOM/DoS - no fix available in base image
|
||||||
"CVE-2025-12084" # Python 3.13 xml.dom.minidom quadratic algorithm - no fix available in base image
|
"CVE-2025-12084" # Python 3.13 xml.dom.minidom quadratic algorithm - no fix available in base image
|
||||||
|
"CVE-2025-60876" # BusyBox wget HTTP request splitting - no fix available in Chainguard Wolfi base image
|
||||||
|
"CVE-2026-0861" # Wolfi glibc still flagged even on 2.42-r5; upstream patched build unavailable yet
|
||||||
|
"CVE-2010-4756" # glibc glob DoS - awaiting patched Wolfi glibc build
|
||||||
|
"CVE-2019-1010022" # glibc stack guard bypass - awaiting patched Wolfi glibc build
|
||||||
|
"CVE-2019-1010023" # glibc ldd remap issue - awaiting patched Wolfi glibc build
|
||||||
|
"CVE-2019-1010024" # glibc ASLR mitigation bypass - awaiting patched Wolfi glibc build
|
||||||
|
"CVE-2019-1010025" # glibc pthread heap address leak - awaiting patched Wolfi glibc build
|
||||||
|
"CVE-2026-22184" # zlib untgz buffer overflow - untgz unused + no fixed Wolfi build yet
|
||||||
|
"GHSA-58pv-8j8x-9vj2" # jaraco.context path traversal - setuptools vendored only (v5.3.0), not used in application code (using v6.1.0+)
|
||||||
|
"GHSA-34x7-hfp2-rc4v" # node-tar hardlink path traversal - not applicable, tar CLI not exposed in application code
|
||||||
|
"GHSA-r6q2-hw4h-h46w" # node-tar not used by application runtime, Linux-only container, not affect by macOS APFS-specific exploit
|
||||||
|
"GHSA-8rrh-rw8j-w5fx" # wheel is from chainguard and will be handled by then TODO: Remove this after Chainguard updates the wheel
|
||||||
|
"CVE-2025-59465" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2025-55131" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2025-59466" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2025-55130" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2025-59467" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2026-21637" # Node only used for Admin UI build/prisma
|
||||||
|
"CVE-2025-55132" # Node only used for Admin UI build/prisma
|
||||||
|
"GHSA-hx9q-6w63-j58v" # orjson dumps recursion; allowlisted
|
||||||
|
"CVE-2025-15281" # No fix available yet
|
||||||
|
"CVE-2026-0865" # No fix available yet
|
||||||
|
"CVE-2025-15282" # No fix available yet
|
||||||
|
"CVE-2026-0672" # No fix available yet
|
||||||
|
"CVE-2025-15366" # No fix available yet
|
||||||
|
"CVE-2025-15367" # No fix available yet
|
||||||
|
"CVE-2025-12781" # No fix available yet
|
||||||
|
"CVE-2025-11468" # No fix available yet
|
||||||
|
"CVE-2026-1299" # Python 3.13 email module header injection - not applicable, LiteLLM doesn't use BytesGenerator for email serialization
|
||||||
|
"CVE-2026-0775" # npm cli incorrect permission assignment - no fix available yet, npm is only used at build/prisma-generate time
|
||||||
)
|
)
|
||||||
|
|
||||||
# Build JSON array of allowlisted CVE IDs for jq
|
# Build JSON array of allowlisted CVE IDs for jq
|
||||||
|
|
@ -158,6 +238,9 @@ main() {
|
||||||
install_trivy
|
install_trivy
|
||||||
install_grype
|
install_grype
|
||||||
|
|
||||||
|
# echo "Running secret detection scans..."
|
||||||
|
# run_secret_detection
|
||||||
|
|
||||||
echo "Running filesystem vulnerability scans..."
|
echo "Running filesystem vulnerability scans..."
|
||||||
run_trivy_scans
|
run_trivy_scans
|
||||||
|
|
||||||
|
|
|
||||||
2
cookbook/LiteLLM_PromptLayer.ipynb
vendored
2
cookbook/LiteLLM_PromptLayer.ipynb
vendored
|
|
@ -39,7 +39,7 @@
|
||||||
"import os\n",
|
"import os\n",
|
||||||
"os.environ['OPENAI_API_KEY'] = \"\"\n",
|
"os.environ['OPENAI_API_KEY'] = \"\"\n",
|
||||||
"os.environ['REPLICATE_API_TOKEN'] = \"\"\n",
|
"os.environ['REPLICATE_API_TOKEN'] = \"\"\n",
|
||||||
"os.environ['PROMPTLAYER_API_KEY'] = \"pl_4ea2bb00a4dca1b8a70cebf2e9e11564\"\n",
|
"os.environ['PROMPTLAYER_API_KEY'] = \"test-promptlayer-key-123\"\n",
|
||||||
"\n",
|
"\n",
|
||||||
"# Set Promptlayer as a success callback\n",
|
"# Set Promptlayer as a success callback\n",
|
||||||
"litellm.success_callback =['promptlayer']\n",
|
"litellm.success_callback =['promptlayer']\n",
|
||||||
|
|
|
||||||
|
|
@ -1,21 +1,10 @@
|
||||||
{
|
{
|
||||||
"nbformat": 4,
|
|
||||||
"nbformat_minor": 0,
|
|
||||||
"metadata": {
|
|
||||||
"colab": {
|
|
||||||
"provenance": []
|
|
||||||
},
|
|
||||||
"kernelspec": {
|
|
||||||
"name": "python3",
|
|
||||||
"display_name": "Python 3"
|
|
||||||
},
|
|
||||||
"language_info": {
|
|
||||||
"name": "python"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"cells": [
|
"cells": [
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "kccfk0mHZ4Ad"
|
||||||
|
},
|
||||||
"source": [
|
"source": [
|
||||||
"# Migrating to LiteLLM Proxy from OpenAI/Azure OpenAI\n",
|
"# Migrating to LiteLLM Proxy from OpenAI/Azure OpenAI\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -32,29 +21,26 @@
|
||||||
"To pass provider-specific args, [go here](https://docs.litellm.ai/docs/completion/provider_specific_params#proxy-usage)\n",
|
"To pass provider-specific args, [go here](https://docs.litellm.ai/docs/completion/provider_specific_params#proxy-usage)\n",
|
||||||
"\n",
|
"\n",
|
||||||
"To drop unsupported params (E.g. frequency_penalty for bedrock with librechat), [go here](https://docs.litellm.ai/docs/completion/drop_params#openai-proxy-usage)\n"
|
"To drop unsupported params (E.g. frequency_penalty for bedrock with librechat), [go here](https://docs.litellm.ai/docs/completion/drop_params#openai-proxy-usage)\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "kccfk0mHZ4Ad"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "nmSClzCPaGH6"
|
||||||
|
},
|
||||||
"source": [
|
"source": [
|
||||||
"## /chat/completion\n",
|
"## /chat/completion\n",
|
||||||
"\n"
|
"\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "nmSClzCPaGH6"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### OpenAI Python SDK"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "_vqcjwOVaKpO"
|
"id": "_vqcjwOVaKpO"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### OpenAI Python SDK"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
|
@ -94,15 +80,20 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"## Function Calling"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "AqkyKk9Scxgj"
|
"id": "AqkyKk9Scxgj"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"## Function Calling"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "wDg10VqLczE1"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"from openai import OpenAI\n",
|
"from openai import OpenAI\n",
|
||||||
"client = OpenAI(\n",
|
"client = OpenAI(\n",
|
||||||
|
|
@ -139,24 +130,24 @@
|
||||||
")\n",
|
")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(completion)\n"
|
"print(completion)\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "wDg10VqLczE1"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Azure OpenAI Python SDK"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "YYoxLloSaNWW"
|
"id": "YYoxLloSaNWW"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Azure OpenAI Python SDK"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "yA1XcgowaSRy"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"import openai\n",
|
"import openai\n",
|
||||||
"client = openai.AzureOpenAI(\n",
|
"client = openai.AzureOpenAI(\n",
|
||||||
|
|
@ -184,24 +175,24 @@
|
||||||
")\n",
|
")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(response)"
|
"print(response)"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "yA1XcgowaSRy"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Langchain Python"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "yl9qhDvnaTpL"
|
"id": "yl9qhDvnaTpL"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Langchain Python"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "5MUZgSquaW5t"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"from langchain.chat_models import ChatOpenAI\n",
|
"from langchain.chat_models import ChatOpenAI\n",
|
||||||
"from langchain.prompts.chat import (\n",
|
"from langchain.prompts.chat import (\n",
|
||||||
|
|
@ -239,24 +230,22 @@
|
||||||
"response = chat(messages)\n",
|
"response = chat(messages)\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(response)"
|
"print(response)"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "5MUZgSquaW5t"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Curl"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "B9eMgnULbRaz"
|
"id": "B9eMgnULbRaz"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Curl"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "VWCCk5PFcmhS"
|
||||||
|
},
|
||||||
"source": [
|
"source": [
|
||||||
"\n",
|
"\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -280,22 +269,24 @@
|
||||||
"}'\n",
|
"}'\n",
|
||||||
"```\n",
|
"```\n",
|
||||||
"\n"
|
"\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "VWCCk5PFcmhS"
|
|
||||||
}
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### LlamaIndex"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "drBAm2e1b6xe"
|
"id": "drBAm2e1b6xe"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### LlamaIndex"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "d0bZcv8fb9mL"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"import os, dotenv\n",
|
"import os, dotenv\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -326,24 +317,24 @@
|
||||||
"query_engine = index.as_query_engine()\n",
|
"query_engine = index.as_query_engine()\n",
|
||||||
"response = query_engine.query(\"What did the author do growing up?\")\n",
|
"response = query_engine.query(\"What did the author do growing up?\")\n",
|
||||||
"print(response)\n"
|
"print(response)\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "d0bZcv8fb9mL"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Langchain JS"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "xypvNdHnb-Yy"
|
"id": "xypvNdHnb-Yy"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Langchain JS"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "R55mK2vCcBN2"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"import { ChatOpenAI } from \"@langchain/openai\";\n",
|
"import { ChatOpenAI } from \"@langchain/openai\";\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -359,24 +350,24 @@
|
||||||
"const message = await model.invoke(\"Hi there!\");\n",
|
"const message = await model.invoke(\"Hi there!\");\n",
|
||||||
"\n",
|
"\n",
|
||||||
"console.log(message);\n"
|
"console.log(message);\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "R55mK2vCcBN2"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### OpenAI JS"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "nC4bLifCcCiW"
|
"id": "nC4bLifCcCiW"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### OpenAI JS"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "MICH8kIMcFpg"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"const { OpenAI } = require('openai');\n",
|
"const { OpenAI } = require('openai');\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -398,24 +389,24 @@
|
||||||
"}\n",
|
"}\n",
|
||||||
"\n",
|
"\n",
|
||||||
"main();\n"
|
"main();\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "MICH8kIMcFpg"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Anthropic SDK"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "D1Q07pEAcGTb"
|
"id": "D1Q07pEAcGTb"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Anthropic SDK"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "qBjFcAvgcI3t"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"import os\n",
|
"import os\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -423,7 +414,7 @@
|
||||||
"\n",
|
"\n",
|
||||||
"client = Anthropic(\n",
|
"client = Anthropic(\n",
|
||||||
" base_url=\"http://localhost:4000\", # proxy endpoint\n",
|
" base_url=\"http://localhost:4000\", # proxy endpoint\n",
|
||||||
" api_key=\"sk-s4xN1IiLTCytwtZFJaYQrA\", # litellm proxy virtual key\n",
|
" api_key=\"sk-test-proxy-key-123\", # litellm proxy virtual key (example)\n",
|
||||||
")\n",
|
")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"message = client.messages.create(\n",
|
"message = client.messages.create(\n",
|
||||||
|
|
@ -437,33 +428,33 @@
|
||||||
" model=\"claude-3-opus-20240229\",\n",
|
" model=\"claude-3-opus-20240229\",\n",
|
||||||
")\n",
|
")\n",
|
||||||
"print(message.content)"
|
"print(message.content)"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "qBjFcAvgcI3t"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"## /embeddings"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "dFAR4AJGcONI"
|
"id": "dFAR4AJGcONI"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"## /embeddings"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### OpenAI Python SDK"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "lgNoM281cRzR"
|
"id": "lgNoM281cRzR"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### OpenAI Python SDK"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "NY3DJhPfcQhA"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"import openai\n",
|
"import openai\n",
|
||||||
"from openai import OpenAI\n",
|
"from openai import OpenAI\n",
|
||||||
|
|
@ -478,24 +469,24 @@
|
||||||
")\n",
|
")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(response)\n"
|
"print(response)\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "NY3DJhPfcQhA"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Langchain Embeddings"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "hmbg-DW6cUZs"
|
"id": "hmbg-DW6cUZs"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Langchain Embeddings"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {
|
||||||
|
"id": "lX2S8Nl1cWVP"
|
||||||
|
},
|
||||||
|
"outputs": [],
|
||||||
"source": [
|
"source": [
|
||||||
"from langchain.embeddings import OpenAIEmbeddings\n",
|
"from langchain.embeddings import OpenAIEmbeddings\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -526,24 +517,22 @@
|
||||||
"\n",
|
"\n",
|
||||||
"print(f\"TITAN EMBEDDINGS\")\n",
|
"print(f\"TITAN EMBEDDINGS\")\n",
|
||||||
"print(query_result[:5])"
|
"print(query_result[:5])"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "lX2S8Nl1cWVP"
|
|
||||||
},
|
|
||||||
"execution_count": null,
|
|
||||||
"outputs": []
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"source": [
|
|
||||||
"### Curl Request"
|
|
||||||
],
|
|
||||||
"metadata": {
|
"metadata": {
|
||||||
"id": "oqGbWBCQcYfd"
|
"id": "oqGbWBCQcYfd"
|
||||||
}
|
},
|
||||||
|
"source": [
|
||||||
|
"### Curl Request"
|
||||||
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
|
"metadata": {
|
||||||
|
"id": "7rkIMV9LcdwQ"
|
||||||
|
},
|
||||||
"source": [
|
"source": [
|
||||||
"\n",
|
"\n",
|
||||||
"\n",
|
"\n",
|
||||||
|
|
@ -556,10 +545,21 @@
|
||||||
" }'\n",
|
" }'\n",
|
||||||
"```\n",
|
"```\n",
|
||||||
"\n"
|
"\n"
|
||||||
],
|
]
|
||||||
"metadata": {
|
|
||||||
"id": "7rkIMV9LcdwQ"
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
]
|
],
|
||||||
}
|
"metadata": {
|
||||||
|
"colab": {
|
||||||
|
"provenance": []
|
||||||
|
},
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"name": "python"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 0
|
||||||
|
}
|
||||||
|
|
|
||||||
295
cookbook/ai_coding_tool_guides/claude_code_quickstart/guide.md
Normal file
295
cookbook/ai_coding_tool_guides/claude_code_quickstart/guide.md
Normal file
|
|
@ -0,0 +1,295 @@
|
||||||
|
# Claude Code with LiteLLM Quickstart
|
||||||
|
|
||||||
|
This guide shows how to call Claude models (and any LiteLLM-supported model) through LiteLLM proxy from Claude Code.
|
||||||
|
|
||||||
|
> **Note:** This integration is based on [Anthropic's official LiteLLM configuration documentation](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration). It allows you to use any LiteLLM supported model through Claude Code with centralized authentication, usage tracking, and cost controls.
|
||||||
|
|
||||||
|
## Video Walkthrough
|
||||||
|
|
||||||
|
Watch the full tutorial: https://www.loom.com/embed/3c17d683cdb74d36a3698763cc558f56
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
|
||||||
|
- API keys for your chosen providers
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
First, install LiteLLM with proxy support:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install 'litellm[proxy]'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Step 1: Setup config.yaml
|
||||||
|
|
||||||
|
Create a secure configuration using environment variables:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
# Claude models
|
||||||
|
- model_name: claude-3-5-sonnet-20241022
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-3-5-sonnet-20241022
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
- model_name: claude-3-5-haiku-20241022
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-3-5-haiku-20241022
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
|
||||||
|
litellm_settings:
|
||||||
|
master_key: os.environ/LITELLM_MASTER_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
Set your environment variables:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export ANTHROPIC_API_KEY="your-anthropic-api-key"
|
||||||
|
export LITELLM_MASTER_KEY="sk-1234567890" # Generate a secure key
|
||||||
|
```
|
||||||
|
|
||||||
|
## Step 2: Start Proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
|
||||||
|
# RUNNING on http://0.0.0.0:4000
|
||||||
|
```
|
||||||
|
|
||||||
|
## Step 3: Verify Setup
|
||||||
|
|
||||||
|
Test that your proxy is working correctly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -X POST http://0.0.0.0:4000/v1/messages \
|
||||||
|
-H "Authorization: Bearer $LITELLM_MASTER_KEY" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "claude-3-5-sonnet-20241022",
|
||||||
|
"max_tokens": 1000,
|
||||||
|
"messages": [{"role": "user", "content": "What is the capital of France?"}]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Step 4: Configure Claude Code
|
||||||
|
|
||||||
|
### Method 1: Unified Endpoint (Recommended)
|
||||||
|
|
||||||
|
Configure Claude Code to use LiteLLM's unified endpoint. Either a virtual key or master key can be used here:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000"
|
||||||
|
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
|
||||||
|
```
|
||||||
|
|
||||||
|
> **Tip:** LITELLM_MASTER_KEY gives Claude access to all proxy models, whereas a virtual key would be limited to the models set in the UI.
|
||||||
|
|
||||||
|
### Method 2: Provider-specific Pass-through Endpoint
|
||||||
|
|
||||||
|
Alternatively, use the Anthropic pass-through endpoint:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export ANTHROPIC_BASE_URL="http://0.0.0.0:4000/anthropic"
|
||||||
|
export ANTHROPIC_AUTH_TOKEN="$LITELLM_MASTER_KEY"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Step 5: Use Claude Code
|
||||||
|
|
||||||
|
### Choosing Your Model
|
||||||
|
|
||||||
|
You have two options for specifying which model Claude Code uses:
|
||||||
|
|
||||||
|
#### Option 1: Command Line / Session Model Selection
|
||||||
|
|
||||||
|
Specify the model directly when starting Claude Code or during a session:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Specify model at startup
|
||||||
|
claude --model claude-3-5-sonnet-20241022
|
||||||
|
|
||||||
|
# Or change model during a session
|
||||||
|
/model claude-3-5-haiku-20241022
|
||||||
|
```
|
||||||
|
|
||||||
|
This method uses the exact model you specify.
|
||||||
|
|
||||||
|
#### Option 2: Environment Variables
|
||||||
|
|
||||||
|
Configure default models using environment variables:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Tell Claude Code which models to use by default
|
||||||
|
export ANTHROPIC_DEFAULT_SONNET_MODEL=claude-3-5-sonnet-20241022
|
||||||
|
export ANTHROPIC_DEFAULT_HAIKU_MODEL=claude-3-5-haiku-20241022
|
||||||
|
export ANTHROPIC_DEFAULT_OPUS_MODEL=claude-opus-3-5-20240229
|
||||||
|
|
||||||
|
claude # Will use the models specified above
|
||||||
|
```
|
||||||
|
|
||||||
|
**Note:** Claude Code may cache the model from a previous session. If environment variables don't take effect, use Option 1 to explicitly set the model.
|
||||||
|
|
||||||
|
**Important:** The `model_name` in your LiteLLM config must match what Claude Code requests (either from env vars or command line).
|
||||||
|
|
||||||
|
### Using 1M Context Window
|
||||||
|
|
||||||
|
Claude Code supports extended context (1 million tokens) using the `[1m]` suffix with Claude 4+ models:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Use Sonnet 4.5 with 1M context (requires quotes for shell)
|
||||||
|
claude --model 'claude-sonnet-4-5-20250929[1m]'
|
||||||
|
|
||||||
|
# Inside a Claude Code session (no quotes needed)
|
||||||
|
/model claude-sonnet-4-5-20250929[1m]
|
||||||
|
```
|
||||||
|
|
||||||
|
**Important:** When using `--model` with `[1m]` in the shell, you must use quotes to prevent the shell from interpreting the brackets.
|
||||||
|
|
||||||
|
Alternatively, set as default with environment variables:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export ANTHROPIC_DEFAULT_SONNET_MODEL='claude-sonnet-4-5-20250929[1m]'
|
||||||
|
claude
|
||||||
|
```
|
||||||
|
|
||||||
|
**How it works:**
|
||||||
|
- Claude Code strips the `[1m]` suffix before sending to LiteLLM
|
||||||
|
- Claude Code automatically adds the header `anthropic-beta: context-1m-2025-08-07`
|
||||||
|
- Your LiteLLM config should **NOT** include `[1m]` in model names
|
||||||
|
|
||||||
|
**Verify 1M context is active:**
|
||||||
|
```bash
|
||||||
|
/context
|
||||||
|
# Should show: 21k/1000k tokens (2%)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Pricing:** Models using 1M context have different pricing. Input tokens above 200k are charged at a higher rate.
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
Common issues and solutions:
|
||||||
|
|
||||||
|
**Claude Code not connecting:**
|
||||||
|
- Verify your proxy is running: `curl http://0.0.0.0:4000/health`
|
||||||
|
- Check that `ANTHROPIC_BASE_URL` is set correctly
|
||||||
|
- Ensure your `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
|
||||||
|
|
||||||
|
**Authentication errors:**
|
||||||
|
- Verify your environment variables are set: `echo $LITELLM_MASTER_KEY`
|
||||||
|
- Check that your API keys are valid and have sufficient credits
|
||||||
|
- Ensure the `ANTHROPIC_AUTH_TOKEN` matches your LiteLLM master key
|
||||||
|
|
||||||
|
**Model not found:**
|
||||||
|
- Check what model Claude Code is requesting in LiteLLM logs
|
||||||
|
- Ensure your `config.yaml` has a matching `model_name` entry
|
||||||
|
- If using environment variables, verify they're set: `echo $ANTHROPIC_DEFAULT_SONNET_MODEL`
|
||||||
|
|
||||||
|
**1M context not working (showing 200k instead of 1000k):**
|
||||||
|
- Verify you're using the `[1m]` suffix: `/model your-model-name[1m]`
|
||||||
|
- Check LiteLLM logs for the header `context-1m-2025-08-07` in the request
|
||||||
|
- Ensure your model supports 1M context (only certain Claude models do)
|
||||||
|
- Your LiteLLM config should **NOT** include `[1m]` in the `model_name`
|
||||||
|
|
||||||
|
## Using Multiple Models and Providers
|
||||||
|
|
||||||
|
You can configure LiteLLM to route to any supported provider. Here's an example with multiple providers:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
# OpenAI models
|
||||||
|
- model_name: codex-mini
|
||||||
|
litellm_params:
|
||||||
|
model: openai/codex-mini
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
api_base: https://api.openai.com/v1
|
||||||
|
|
||||||
|
- model_name: o3-pro
|
||||||
|
litellm_params:
|
||||||
|
model: openai/o3-pro
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
api_base: https://api.openai.com/v1
|
||||||
|
|
||||||
|
- model_name: gpt-4o
|
||||||
|
litellm_params:
|
||||||
|
model: openai/gpt-4o
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
api_base: https://api.openai.com/v1
|
||||||
|
|
||||||
|
# Anthropic models
|
||||||
|
- model_name: claude-3-5-sonnet-20241022
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-3-5-sonnet-20241022
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
- model_name: claude-3-5-haiku-20241022
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-3-5-haiku-20241022
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
# AWS Bedrock
|
||||||
|
- model_name: claude-bedrock
|
||||||
|
litellm_params:
|
||||||
|
model: bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0
|
||||||
|
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||||
|
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||||
|
aws_region_name: us-east-1
|
||||||
|
|
||||||
|
litellm_settings:
|
||||||
|
master_key: os.environ/LITELLM_MASTER_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
**Note:** The `model_name` can be anything you choose. Claude Code will request whatever model you specify (via env vars or command line), and LiteLLM will route to the `model` configured in `litellm_params`.
|
||||||
|
|
||||||
|
Switch between models seamlessly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Use environment variables to set defaults
|
||||||
|
export ANTHROPIC_DEFAULT_SONNET_MODEL=claude-3-5-sonnet-20241022
|
||||||
|
export ANTHROPIC_DEFAULT_HAIKU_MODEL=claude-3-5-haiku-20241022
|
||||||
|
|
||||||
|
# Or specify directly
|
||||||
|
claude --model claude-3-5-sonnet-20241022 # Complex reasoning
|
||||||
|
claude --model claude-3-5-haiku-20241022 # Fast responses
|
||||||
|
claude --model claude-bedrock # Bedrock deployment
|
||||||
|
```
|
||||||
|
|
||||||
|
## Default Models Used by Claude Code
|
||||||
|
|
||||||
|
If you **don't** set environment variables, Claude Code uses these default model names:
|
||||||
|
|
||||||
|
| Purpose | Default Model Name (v2.1.14) |
|
||||||
|
|---------|------------------------------|
|
||||||
|
| Main model | `claude-sonnet-4-5-20250929` |
|
||||||
|
| Light tasks (subagents, summaries) | `claude-haiku-4-5-20251001` |
|
||||||
|
| Planning mode | `claude-opus-4-5-20251101` |
|
||||||
|
|
||||||
|
Your LiteLLM config should include these model names if you want Claude Code to work without setting environment variables:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-sonnet-4-5-20250929
|
||||||
|
litellm_params:
|
||||||
|
# Can be any provider - Anthropic, Bedrock, Vertex AI, etc.
|
||||||
|
model: anthropic/claude-sonnet-4-5-20250929
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
- model_name: claude-haiku-4-5-20251001
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-haiku-4-5-20251001
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
|
- model_name: claude-opus-4-5-20251101
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-opus-4-5-20251101
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
**Warning:** These default model names may change with new Claude Code versions. Check LiteLLM proxy logs for "model not found" errors to identify what Claude Code is requesting.
|
||||||
|
|
||||||
|
## Additional Resources
|
||||||
|
|
||||||
|
- [LiteLLM Documentation](https://docs.litellm.ai/)
|
||||||
|
- [Claude Code Documentation](https://docs.anthropic.com/en/docs/claude-code/overview)
|
||||||
|
- [Anthropic's LiteLLM Configuration Guide](https://docs.anthropic.com/en/docs/claude-code/llm-gateway#litellm-configuration)
|
||||||
|
|
||||||
134
cookbook/ai_coding_tool_guides/index.json
Normal file
134
cookbook/ai_coding_tool_guides/index.json
Normal file
|
|
@ -0,0 +1,134 @@
|
||||||
|
[{
|
||||||
|
"title": "Claude Code Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using Claude Code with LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/claude_responses_api",
|
||||||
|
"date": "2026-01-15",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"LiteLLM"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Claude Code with MCPs",
|
||||||
|
"description": "This is a guide to using Claude Code with MCPs via LiteLLM Proxy.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/claude_mcp",
|
||||||
|
"date": "2026-01-15",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"LiteLLM",
|
||||||
|
"MCP"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Claude Code with Non-Anthropic Models",
|
||||||
|
"description": "This is a guide to using Claude Code with non-Anthropic models via LiteLLM Proxy.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/claude_non_anthropic_models",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"LiteLLM",
|
||||||
|
"OpenAI",
|
||||||
|
"Gemini"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Cursor Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using Cursor with LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/cursor_integration",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Cursor",
|
||||||
|
"LiteLLM",
|
||||||
|
"Quickstart"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Github Copilot Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using Github Copilot with LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/github_copilot_integration",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Github Copilot",
|
||||||
|
"LiteLLM",
|
||||||
|
"Quickstart"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "LiteLLM Gemini CLI Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using LiteLLM Gemini CLI.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/litellm_gemini_cli",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Gemini CLI",
|
||||||
|
"Gemini",
|
||||||
|
"LiteLLM",
|
||||||
|
"Quickstart"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "OpenAI Codex CLI Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using OpenAI Codex CLI.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/openai_codex",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"OpenAI Codex CLI",
|
||||||
|
"OpenAI",
|
||||||
|
"LiteLLM",
|
||||||
|
"Quickstart"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "OpenWebUI Quickstart",
|
||||||
|
"description": "This is a quickstart guide to using OpenWebUI with LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/openweb_ui",
|
||||||
|
"date": "2026-01-16",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"OpenWebUI",
|
||||||
|
"LiteLLM",
|
||||||
|
"Quickstart"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "AI Coding Tool Usage Tracking",
|
||||||
|
"description": "This is a guide to tracking usage for AI coding tools monitor the use of Claude Code , Google Antigravity, OpenAI Codex, Roo Code etc. through LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/cost_tracking_coding",
|
||||||
|
"date": "2026-01-17",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"Gemini CLI",
|
||||||
|
"OpenAI Codex",
|
||||||
|
"LiteLLM"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Use Web Search with Claude Code (across Bedrock/OpenAI/Gemini/etc.)",
|
||||||
|
"description": "This is a guide for using Web Search with Claude Code via LiteLLM.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/claude_code_websearch",
|
||||||
|
"date": "2026-01-17",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"LiteLLM",
|
||||||
|
"Web Search"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"title": "Track Claude Code Usage per user via Custom Headers",
|
||||||
|
"description": "This is a guide for tracking claude code user usage by passing a customer ID header.",
|
||||||
|
"url": "https://docs.litellm.ai/docs/tutorials/claude_code_customer_tracking",
|
||||||
|
"date": "2026-01-17",
|
||||||
|
"version": "1.0.0",
|
||||||
|
"tags": [
|
||||||
|
"Claude Code",
|
||||||
|
"LiteLLM"
|
||||||
|
]
|
||||||
|
}]
|
||||||
144
cookbook/anthropic_agent_sdk/README.md
Normal file
144
cookbook/anthropic_agent_sdk/README.md
Normal file
|
|
@ -0,0 +1,144 @@
|
||||||
|
# Claude Agent SDK with LiteLLM Gateway
|
||||||
|
|
||||||
|
A simple example showing how to use Claude's Agent SDK with LiteLLM as a proxy. This lets you use any LLM provider (OpenAI, Bedrock, Azure, etc.) through the Agent SDK.
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
### 1. Install dependencies
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install anthropic claude-agent-sdk litellm
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Start LiteLLM proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Simple start with Claude
|
||||||
|
litellm --model claude-sonnet-4-20250514
|
||||||
|
|
||||||
|
# Or with a config file
|
||||||
|
litellm --config config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3. Run the chat
|
||||||
|
|
||||||
|
**Basic Agent (no MCP):**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
**Agent with MCP (DeepWiki2 for research):**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python agent_with_mcp.py
|
||||||
|
```
|
||||||
|
|
||||||
|
If MCP connection fails, you can disable it:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
USE_MCP=false python agent_with_mcp.py
|
||||||
|
```
|
||||||
|
|
||||||
|
That's it! You can now chat with the agent in your terminal.
|
||||||
|
|
||||||
|
### Chat Commands
|
||||||
|
|
||||||
|
While chatting, you can use these commands:
|
||||||
|
- `models` - List all available models (fetched from your LiteLLM proxy)
|
||||||
|
- `model` - Switch to a different model
|
||||||
|
- `clear` - Start a new conversation
|
||||||
|
- `quit` or `exit` - End the chat
|
||||||
|
|
||||||
|
The chat automatically fetches available models from your LiteLLM proxy's `/models` endpoint, so you'll always see what's currently configured.
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
Set these environment variables if needed:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export LITELLM_PROXY_URL="http://localhost:4000"
|
||||||
|
export LITELLM_API_KEY="sk-1234"
|
||||||
|
export LITELLM_MODEL="bedrock-claude-sonnet-4.5"
|
||||||
|
```
|
||||||
|
|
||||||
|
Or just use the defaults - it'll connect to `http://localhost:4000` by default.
|
||||||
|
|
||||||
|
## Files
|
||||||
|
|
||||||
|
- `main.py` - Basic interactive agent without MCP
|
||||||
|
- `agent_with_mcp.py` - Agent with MCP server integration (DeepWiki2)
|
||||||
|
- `common.py` - Shared utilities and functions
|
||||||
|
- `config.example.yaml` - Example LiteLLM configuration
|
||||||
|
- `requirements.txt` - Python dependencies
|
||||||
|
|
||||||
|
## Example Config File
|
||||||
|
|
||||||
|
If you want to use multiple models, create a `config.yaml` (see `config.example.yaml`):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: bedrock-claude-sonnet-4
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
|
||||||
|
- model_name: bedrock-claude-sonnet-4.5
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
```
|
||||||
|
|
||||||
|
Then start LiteLLM with: `litellm --config config.yaml`
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
The key is pointing the Agent SDK to LiteLLM instead of directly to Anthropic:
|
||||||
|
|
||||||
|
```python
|
||||||
|
# Point to LiteLLM gateway (not Anthropic)
|
||||||
|
os.environ["ANTHROPIC_BASE_URL"] = "http://localhost:4000"
|
||||||
|
os.environ["ANTHROPIC_API_KEY"] = "sk-1234" # Your LiteLLM key
|
||||||
|
|
||||||
|
# Use any model configured in LiteLLM
|
||||||
|
options = ClaudeAgentOptions(
|
||||||
|
model="bedrock-claude-sonnet-4", # or gpt-4, or anything else
|
||||||
|
system_prompt="You are a helpful assistant.",
|
||||||
|
max_turns=50,
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
Note: Don't add `/anthropic` to the base URL - LiteLLM handles the routing automatically.
|
||||||
|
|
||||||
|
## Why Use This?
|
||||||
|
|
||||||
|
- **Switch providers easily**: Use the same code with OpenAI, Bedrock, Azure, etc.
|
||||||
|
- **Cost tracking**: LiteLLM tracks spending across all your agent conversations
|
||||||
|
- **Rate limiting**: Set budgets and limits on your agent usage
|
||||||
|
- **Load balancing**: Distribute requests across multiple API keys or regions
|
||||||
|
- **Fallbacks**: Automatically retry with a different model if one fails
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
**Connection errors?**
|
||||||
|
- Make sure LiteLLM is running: `litellm --model your-model`
|
||||||
|
- Check the URL is correct (default: `http://localhost:4000`)
|
||||||
|
|
||||||
|
**Authentication errors?**
|
||||||
|
- Verify your LiteLLM API key is correct
|
||||||
|
- Make sure the model is configured in your LiteLLM setup
|
||||||
|
|
||||||
|
**Model not found?**
|
||||||
|
- Check the model name matches what's in your LiteLLM config
|
||||||
|
- Run `litellm --model your-model` to test it works
|
||||||
|
|
||||||
|
**Agent with MCP stuck or failing?**
|
||||||
|
- The MCP server might not be available at `http://localhost:4000/mcp/deepwiki2`
|
||||||
|
- Try disabling MCP: `USE_MCP=false python agent_with_mcp.py`
|
||||||
|
- Or use the basic agent: `python main.py`
|
||||||
|
|
||||||
|
## Learn More
|
||||||
|
|
||||||
|
- [LiteLLM Docs](https://docs.litellm.ai/)
|
||||||
|
- [Claude Agent SDK](https://github.com/anthropics/anthropic-agent-sdk)
|
||||||
|
- [LiteLLM Proxy Guide](https://docs.litellm.ai/docs/proxy/quick_start)
|
||||||
140
cookbook/anthropic_agent_sdk/agent_with_mcp.py
Normal file
140
cookbook/anthropic_agent_sdk/agent_with_mcp.py
Normal file
|
|
@ -0,0 +1,140 @@
|
||||||
|
"""
|
||||||
|
Interactive Claude Agent SDK CLI with MCP Support
|
||||||
|
|
||||||
|
This example demonstrates an interactive CLI chat with the Anthropic Agent SDK using LiteLLM as a proxy,
|
||||||
|
with MCP (Model Context Protocol) server integration for enhanced capabilities.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
from claude_agent_sdk import ClaudeSDKClient, ClaudeAgentOptions
|
||||||
|
from common import (
|
||||||
|
Config,
|
||||||
|
fetch_available_models,
|
||||||
|
setup_litellm_env,
|
||||||
|
print_header,
|
||||||
|
handle_model_list,
|
||||||
|
handle_model_switch,
|
||||||
|
stream_response,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def interactive_chat_with_mcp():
|
||||||
|
"""
|
||||||
|
Interactive CLI chat with the agent and MCP server
|
||||||
|
"""
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
# Configure Anthropic SDK to point to LiteLLM gateway
|
||||||
|
litellm_base_url = setup_litellm_env(config)
|
||||||
|
|
||||||
|
# Fetch available models from proxy
|
||||||
|
available_models = await fetch_available_models(litellm_base_url, config.LITELLM_API_KEY)
|
||||||
|
|
||||||
|
current_model = config.LITELLM_MODEL
|
||||||
|
|
||||||
|
# MCP server configuration
|
||||||
|
mcp_server_url = f"{litellm_base_url}/mcp/deepwiki2"
|
||||||
|
use_mcp = os.getenv("USE_MCP", "true").lower() == "true"
|
||||||
|
|
||||||
|
if not use_mcp:
|
||||||
|
print("⚠️ MCP disabled via USE_MCP=false")
|
||||||
|
|
||||||
|
print_header(litellm_base_url, current_model, has_mcp=use_mcp)
|
||||||
|
|
||||||
|
while True:
|
||||||
|
# Configure agent options
|
||||||
|
if use_mcp:
|
||||||
|
try:
|
||||||
|
# Try with MCP server (HTTP transport)
|
||||||
|
# Using McpHttpServerConfig format from Agent SDK
|
||||||
|
options = ClaudeAgentOptions(
|
||||||
|
system_prompt="You are a helpful AI assistant with access to DeepWiki for research. Be concise, accurate, and friendly.",
|
||||||
|
model=current_model,
|
||||||
|
max_turns=50,
|
||||||
|
mcp_servers={
|
||||||
|
"deepwiki2": {
|
||||||
|
"type": "http",
|
||||||
|
"url": mcp_server_url,
|
||||||
|
"headers": {
|
||||||
|
"Authorization": f"Bearer {config.LITELLM_API_KEY}"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"⚠️ Warning: Could not configure MCP server: {e}")
|
||||||
|
print("Continuing without MCP...\n")
|
||||||
|
use_mcp = False
|
||||||
|
options = ClaudeAgentOptions(
|
||||||
|
system_prompt="You are a helpful AI assistant. Be concise, accurate, and friendly.",
|
||||||
|
model=current_model,
|
||||||
|
max_turns=50,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Without MCP
|
||||||
|
options = ClaudeAgentOptions(
|
||||||
|
system_prompt="You are a helpful AI assistant. Be concise, accurate, and friendly.",
|
||||||
|
model=current_model,
|
||||||
|
max_turns=50,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create agent client
|
||||||
|
try:
|
||||||
|
async with ClaudeSDKClient(options=options) as client:
|
||||||
|
conversation_active = True
|
||||||
|
|
||||||
|
while conversation_active:
|
||||||
|
# Get user input
|
||||||
|
try:
|
||||||
|
user_input = input("\n👤 You: ").strip()
|
||||||
|
except (EOFError, KeyboardInterrupt):
|
||||||
|
print("\n\n👋 Goodbye!")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Handle commands
|
||||||
|
if user_input.lower() in ['quit', 'exit']:
|
||||||
|
print("\n👋 Goodbye!")
|
||||||
|
return
|
||||||
|
|
||||||
|
if user_input.lower() == 'clear':
|
||||||
|
print("\n🔄 Starting new conversation...\n")
|
||||||
|
conversation_active = False
|
||||||
|
continue
|
||||||
|
|
||||||
|
if user_input.lower() == 'models':
|
||||||
|
handle_model_list(available_models, current_model)
|
||||||
|
continue
|
||||||
|
|
||||||
|
if user_input.lower() == 'model':
|
||||||
|
new_model, should_restart = handle_model_switch(available_models, current_model)
|
||||||
|
if should_restart:
|
||||||
|
current_model = new_model
|
||||||
|
conversation_active = False
|
||||||
|
continue
|
||||||
|
|
||||||
|
if not user_input:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Stream response from agent
|
||||||
|
await stream_response(client, user_input)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"\n❌ Error creating agent client: {e}")
|
||||||
|
print("This might be an MCP configuration issue. Try running without MCP:")
|
||||||
|
print(" USE_MCP=false python agent_with_mcp.py")
|
||||||
|
print("\nOr use the basic agent:")
|
||||||
|
print(" python main.py")
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Run interactive chat with MCP"""
|
||||||
|
try:
|
||||||
|
asyncio.run(interactive_chat_with_mcp())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n\n👋 Goodbye!")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
160
cookbook/anthropic_agent_sdk/common.py
Normal file
160
cookbook/anthropic_agent_sdk/common.py
Normal file
|
|
@ -0,0 +1,160 @@
|
||||||
|
"""
|
||||||
|
Common utilities for Claude Agent SDK examples
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
|
||||||
|
class Config:
|
||||||
|
"""Configuration for LiteLLM Gateway connection"""
|
||||||
|
|
||||||
|
# LiteLLM proxy URL (default to local instance)
|
||||||
|
LITELLM_PROXY_URL = os.getenv("LITELLM_PROXY_URL", "http://localhost:4000")
|
||||||
|
|
||||||
|
# LiteLLM API key (master key or virtual key)
|
||||||
|
LITELLM_API_KEY = os.getenv("LITELLM_API_KEY", "sk-1234")
|
||||||
|
|
||||||
|
# Model name as configured in LiteLLM (e.g., "bedrock-claude-sonnet-4", "gpt-4", etc.)
|
||||||
|
LITELLM_MODEL = os.getenv("LITELLM_MODEL", "bedrock-claude-sonnet-4.5")
|
||||||
|
|
||||||
|
|
||||||
|
async def fetch_available_models(base_url: str, api_key: str) -> list[str]:
|
||||||
|
"""
|
||||||
|
Fetch available models from LiteLLM proxy /models endpoint
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
async with httpx.AsyncClient() as client:
|
||||||
|
response = await client.get(
|
||||||
|
f"{base_url}/models",
|
||||||
|
headers={"Authorization": f"Bearer {api_key}"},
|
||||||
|
timeout=10.0
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
data = response.json()
|
||||||
|
return [model["id"] for model in data.get("data", [])]
|
||||||
|
except Exception as e:
|
||||||
|
print(f"⚠️ Warning: Could not fetch models from proxy: {e}")
|
||||||
|
print("Using default model list...")
|
||||||
|
# Fallback to default models
|
||||||
|
return [
|
||||||
|
"bedrock-claude-sonnet-3.5",
|
||||||
|
"bedrock-claude-sonnet-4",
|
||||||
|
"bedrock-claude-sonnet-4.5",
|
||||||
|
"bedrock-claude-opus-4.5",
|
||||||
|
"bedrock-nova-premier",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def setup_litellm_env(config: Config):
|
||||||
|
"""
|
||||||
|
Configure environment variables to point Agent SDK to LiteLLM
|
||||||
|
"""
|
||||||
|
litellm_base_url = config.LITELLM_PROXY_URL.rstrip('/')
|
||||||
|
os.environ["ANTHROPIC_BASE_URL"] = litellm_base_url
|
||||||
|
os.environ["ANTHROPIC_API_KEY"] = config.LITELLM_API_KEY
|
||||||
|
return litellm_base_url
|
||||||
|
|
||||||
|
|
||||||
|
def print_header(base_url: str, current_model: str, has_mcp: bool = False):
|
||||||
|
"""
|
||||||
|
Print the chat header
|
||||||
|
"""
|
||||||
|
mcp_indicator = " + MCP" if has_mcp else ""
|
||||||
|
print("=" * 70)
|
||||||
|
print(f"🤖 Claude Agent SDK with LiteLLM Gateway{mcp_indicator} - Interactive Chat")
|
||||||
|
print("=" * 70)
|
||||||
|
print(f"🚀 Connected to: {base_url}")
|
||||||
|
print(f"📦 Current model: {current_model}")
|
||||||
|
if has_mcp:
|
||||||
|
print("🔌 MCP: deepwiki2 enabled")
|
||||||
|
print("\nType your messages below. Commands:")
|
||||||
|
print(" - 'quit' or 'exit' to end the conversation")
|
||||||
|
print(" - 'clear' to start a new conversation")
|
||||||
|
print(" - 'model' to switch models")
|
||||||
|
print(" - 'models' to list available models")
|
||||||
|
print("=" * 70)
|
||||||
|
print()
|
||||||
|
|
||||||
|
|
||||||
|
def handle_model_list(available_models: list[str], current_model: str):
|
||||||
|
"""
|
||||||
|
Display available models
|
||||||
|
"""
|
||||||
|
print("\n📋 Available models:")
|
||||||
|
for i, model in enumerate(available_models, 1):
|
||||||
|
marker = "✓" if model == current_model else " "
|
||||||
|
print(f" {marker} {i}. {model}")
|
||||||
|
|
||||||
|
|
||||||
|
def handle_model_switch(available_models: list[str], current_model: str) -> tuple[str, bool]:
|
||||||
|
"""
|
||||||
|
Handle model switching
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple: (new_model, should_restart_conversation)
|
||||||
|
"""
|
||||||
|
print("\n📋 Select a model:")
|
||||||
|
for i, model in enumerate(available_models, 1):
|
||||||
|
marker = "✓" if model == current_model else " "
|
||||||
|
print(f" {marker} {i}. {model}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
choice = input("\nEnter number (or press Enter to cancel): ").strip()
|
||||||
|
if choice:
|
||||||
|
idx = int(choice) - 1
|
||||||
|
if 0 <= idx < len(available_models):
|
||||||
|
new_model = available_models[idx]
|
||||||
|
print(f"\n✅ Switched to: {new_model}")
|
||||||
|
print("🔄 Starting new conversation with new model...\n")
|
||||||
|
return new_model, True
|
||||||
|
else:
|
||||||
|
print("❌ Invalid choice")
|
||||||
|
except (ValueError, IndexError):
|
||||||
|
print("❌ Invalid input")
|
||||||
|
|
||||||
|
return current_model, False
|
||||||
|
|
||||||
|
|
||||||
|
async def stream_response(client, user_input: str):
|
||||||
|
"""
|
||||||
|
Stream response from the agent
|
||||||
|
"""
|
||||||
|
print("\n🤖 Assistant: ", end='', flush=True)
|
||||||
|
|
||||||
|
try:
|
||||||
|
await client.query(user_input)
|
||||||
|
|
||||||
|
# Show loading indicator
|
||||||
|
print("⏳ thinking...", end='', flush=True)
|
||||||
|
|
||||||
|
# Stream the response
|
||||||
|
first_chunk = True
|
||||||
|
async for msg in client.receive_response():
|
||||||
|
# Clear loading indicator on first message
|
||||||
|
if first_chunk:
|
||||||
|
print("\r🤖 Assistant: ", end='', flush=True)
|
||||||
|
first_chunk = False
|
||||||
|
|
||||||
|
# Handle different message types
|
||||||
|
if hasattr(msg, 'type'):
|
||||||
|
if msg.type == 'content_block_delta':
|
||||||
|
# Streaming text delta
|
||||||
|
if hasattr(msg, 'delta') and hasattr(msg.delta, 'text'):
|
||||||
|
print(msg.delta.text, end='', flush=True)
|
||||||
|
elif msg.type == 'content_block_start':
|
||||||
|
# Start of content block
|
||||||
|
if hasattr(msg, 'content_block') and hasattr(msg.content_block, 'text'):
|
||||||
|
print(msg.content_block.text, end='', flush=True)
|
||||||
|
|
||||||
|
# Fallback to original content handling
|
||||||
|
if hasattr(msg, 'content'):
|
||||||
|
for content_block in msg.content:
|
||||||
|
if hasattr(content_block, 'text'):
|
||||||
|
print(content_block.text, end='', flush=True)
|
||||||
|
|
||||||
|
print() # New line after response
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"\r\n❌ Error: {e}")
|
||||||
|
print("Please check your LiteLLM gateway is running and configured correctly.")
|
||||||
25
cookbook/anthropic_agent_sdk/config.example.yaml
Normal file
25
cookbook/anthropic_agent_sdk/config.example.yaml
Normal file
|
|
@ -0,0 +1,25 @@
|
||||||
|
model_list:
|
||||||
|
- model_name: bedrock-claude-sonnet-3.5
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
|
||||||
|
- model_name: bedrock-claude-sonnet-4
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-sonnet-4-20250514-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
|
||||||
|
- model_name: bedrock-claude-sonnet-4.5
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
|
||||||
|
- model_name: bedrock-claude-opus-4.5
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/us.anthropic.claude-opus-4-5-20251101-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
|
|
||||||
|
- model_name: bedrock-nova-premier
|
||||||
|
litellm_params:
|
||||||
|
model: "bedrock/amazon.nova-premier-v1:0"
|
||||||
|
aws_region_name: "us-east-1"
|
||||||
95
cookbook/anthropic_agent_sdk/main.py
Normal file
95
cookbook/anthropic_agent_sdk/main.py
Normal file
|
|
@ -0,0 +1,95 @@
|
||||||
|
"""
|
||||||
|
Simple Interactive Claude Agent SDK CLI using LiteLLM Gateway
|
||||||
|
|
||||||
|
This example demonstrates an interactive CLI chat with the Anthropic Agent SDK using LiteLLM as a proxy.
|
||||||
|
LiteLLM acts as a unified interface, allowing you to use any LLM provider (OpenAI, Azure, Bedrock, etc.)
|
||||||
|
through the Claude Agent SDK by pointing it to the LiteLLM gateway.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from claude_agent_sdk import ClaudeSDKClient, ClaudeAgentOptions
|
||||||
|
from common import (
|
||||||
|
Config,
|
||||||
|
fetch_available_models,
|
||||||
|
setup_litellm_env,
|
||||||
|
print_header,
|
||||||
|
handle_model_list,
|
||||||
|
handle_model_switch,
|
||||||
|
stream_response,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def interactive_chat():
|
||||||
|
"""
|
||||||
|
Interactive CLI chat with the agent
|
||||||
|
"""
|
||||||
|
config = Config()
|
||||||
|
|
||||||
|
# Configure Anthropic SDK to point to LiteLLM gateway
|
||||||
|
litellm_base_url = setup_litellm_env(config)
|
||||||
|
|
||||||
|
# Fetch available models from proxy
|
||||||
|
available_models = await fetch_available_models(litellm_base_url, config.LITELLM_API_KEY)
|
||||||
|
|
||||||
|
current_model = config.LITELLM_MODEL
|
||||||
|
|
||||||
|
print_header(litellm_base_url, current_model)
|
||||||
|
|
||||||
|
while True:
|
||||||
|
# Configure agent options for each conversation
|
||||||
|
options = ClaudeAgentOptions(
|
||||||
|
system_prompt="You are a helpful AI assistant. Be concise, accurate, and friendly.",
|
||||||
|
model=current_model,
|
||||||
|
max_turns=50,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Create agent client
|
||||||
|
async with ClaudeSDKClient(options=options) as client:
|
||||||
|
conversation_active = True
|
||||||
|
|
||||||
|
while conversation_active:
|
||||||
|
# Get user input
|
||||||
|
try:
|
||||||
|
user_input = input("\n👤 You: ").strip()
|
||||||
|
except (EOFError, KeyboardInterrupt):
|
||||||
|
print("\n\n👋 Goodbye!")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Handle commands
|
||||||
|
if user_input.lower() in ['quit', 'exit']:
|
||||||
|
print("\n👋 Goodbye!")
|
||||||
|
return
|
||||||
|
|
||||||
|
if user_input.lower() == 'clear':
|
||||||
|
print("\n🔄 Starting new conversation...\n")
|
||||||
|
conversation_active = False
|
||||||
|
continue
|
||||||
|
|
||||||
|
if user_input.lower() == 'models':
|
||||||
|
handle_model_list(available_models, current_model)
|
||||||
|
continue
|
||||||
|
|
||||||
|
if user_input.lower() == 'model':
|
||||||
|
new_model, should_restart = handle_model_switch(available_models, current_model)
|
||||||
|
if should_restart:
|
||||||
|
current_model = new_model
|
||||||
|
conversation_active = False
|
||||||
|
continue
|
||||||
|
|
||||||
|
if not user_input:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Stream response from agent
|
||||||
|
await stream_response(client, user_input)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Run interactive chat"""
|
||||||
|
try:
|
||||||
|
asyncio.run(interactive_chat())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n\n👋 Goodbye!")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
2
cookbook/anthropic_agent_sdk/requirements.txt
Normal file
2
cookbook/anthropic_agent_sdk/requirements.txt
Normal file
|
|
@ -0,0 +1,2 @@
|
||||||
|
claude-agent-sdk
|
||||||
|
httpx>=0.27.0
|
||||||
114
cookbook/livekit_agent_sdk/README.md
Normal file
114
cookbook/livekit_agent_sdk/README.md
Normal file
|
|
@ -0,0 +1,114 @@
|
||||||
|
# LiveKit Voice Agent with LiteLLM Gateway
|
||||||
|
|
||||||
|
Simple example showing how to use LiveKit's xAI realtime plugin with LiteLLM as a proxy. This lets you switch between xAI, OpenAI, and Azure realtime APIs without changing your code.
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
### 1. Install dependencies
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install livekit-agents[xai] websockets
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Start LiteLLM proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# With xAI
|
||||||
|
export XAI_API_KEY="your-xai-key"
|
||||||
|
litellm --config config.yaml --port 4000
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3. Run the voice agent
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Type your message and get a voice response from Grok!
|
||||||
|
|
||||||
|
## Configuration
|
||||||
|
|
||||||
|
Set these environment variables if needed:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export LITELLM_PROXY_URL="http://localhost:4000"
|
||||||
|
export LITELLM_API_KEY="sk-1234"
|
||||||
|
export LITELLM_MODEL="grok-voice-agent"
|
||||||
|
```
|
||||||
|
|
||||||
|
Or use the defaults - connects to `http://localhost:4000` by default.
|
||||||
|
|
||||||
|
## Example Config File
|
||||||
|
|
||||||
|
Create a `config.yaml` with your realtime models:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: grok-voice-agent
|
||||||
|
litellm_params:
|
||||||
|
model: xai/grok-2-vision-1212
|
||||||
|
api_key: os.environ/XAI_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: realtime
|
||||||
|
|
||||||
|
- model_name: openai-voice-agent
|
||||||
|
litellm_params:
|
||||||
|
model: gpt-4o-realtime-preview
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: realtime
|
||||||
|
|
||||||
|
general_settings:
|
||||||
|
master_key: sk-1234
|
||||||
|
```
|
||||||
|
|
||||||
|
Then start: `litellm --config config.yaml --port 4000`
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
LiveKit's xAI plugin connects through LiteLLM proxy by setting `base_url`:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from livekit.plugins import xai
|
||||||
|
|
||||||
|
model = xai.realtime.RealtimeModel(
|
||||||
|
voice="ara",
|
||||||
|
api_key="sk-1234", # LiteLLM proxy key
|
||||||
|
base_url="http://localhost:4000", # Point to LiteLLM
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Switching Providers
|
||||||
|
|
||||||
|
Just change the model in your config - no code changes needed:
|
||||||
|
|
||||||
|
**xAI Grok:**
|
||||||
|
```yaml
|
||||||
|
model: xai/grok-2-vision-1212
|
||||||
|
```
|
||||||
|
|
||||||
|
**OpenAI:**
|
||||||
|
```yaml
|
||||||
|
model: gpt-4o-realtime-preview
|
||||||
|
```
|
||||||
|
|
||||||
|
**Azure OpenAI:**
|
||||||
|
```yaml
|
||||||
|
model: azure/gpt-4o-realtime-preview
|
||||||
|
api_base: https://your-endpoint.openai.azure.com/
|
||||||
|
```
|
||||||
|
|
||||||
|
## Why Use LiteLLM?
|
||||||
|
|
||||||
|
- ✅ **Switch providers** without changing agent code
|
||||||
|
- ✅ **Cost tracking** across all voice sessions
|
||||||
|
- ✅ **Rate limiting** and budgets
|
||||||
|
- ✅ **Load balancing** across multiple API keys
|
||||||
|
- ✅ **Fallbacks** to backup models
|
||||||
|
|
||||||
|
## Learn More
|
||||||
|
|
||||||
|
- [LiveKit xAI Realtime Tutorial](/docs/tutorials/livekit_xai_realtime)
|
||||||
|
- [xAI Realtime Docs](/docs/providers/xai_realtime)
|
||||||
|
- [LiveKit Agents Documentation](https://docs.livekit.io/agents/)
|
||||||
|
- [LiteLLM Realtime API](/docs/realtime)
|
||||||
21
cookbook/livekit_agent_sdk/config.example.yaml
Normal file
21
cookbook/livekit_agent_sdk/config.example.yaml
Normal file
|
|
@ -0,0 +1,21 @@
|
||||||
|
model_list:
|
||||||
|
- model_name: grok-voice-agent
|
||||||
|
litellm_params:
|
||||||
|
model: xai/grok-2-vision-1212
|
||||||
|
api_key: os.environ/XAI_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: realtime
|
||||||
|
|
||||||
|
- model_name: openai-voice-agent
|
||||||
|
litellm_params:
|
||||||
|
model: gpt-4o-realtime-preview
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
model_info:
|
||||||
|
mode: realtime
|
||||||
|
|
||||||
|
litellm_settings:
|
||||||
|
drop_params: True
|
||||||
|
telemetry: False
|
||||||
|
|
||||||
|
general_settings:
|
||||||
|
master_key: sk-1234 # Change this to a secure key
|
||||||
112
cookbook/livekit_agent_sdk/main.py
Normal file
112
cookbook/livekit_agent_sdk/main.py
Normal file
|
|
@ -0,0 +1,112 @@
|
||||||
|
"""
|
||||||
|
Simple xAI Voice Agent using LiveKit SDK with LiteLLM Gateway
|
||||||
|
|
||||||
|
This example shows how to use LiveKit's xAI realtime plugin through LiteLLM proxy.
|
||||||
|
LiteLLM acts as a unified interface, allowing you to switch between xAI, OpenAI,
|
||||||
|
and Azure realtime APIs without changing your agent code.
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import websockets
|
||||||
|
|
||||||
|
# Configuration
|
||||||
|
PROXY_URL = os.getenv("LITELLM_PROXY_URL", "http://localhost:4000")
|
||||||
|
API_KEY = os.getenv("LITELLM_API_KEY", "sk-1234")
|
||||||
|
MODEL = os.getenv("LITELLM_MODEL", "grok-voice-agent")
|
||||||
|
|
||||||
|
|
||||||
|
async def run_voice_agent():
|
||||||
|
"""
|
||||||
|
Simple voice agent that:
|
||||||
|
1. Connects to xAI realtime API through LiteLLM proxy
|
||||||
|
2. Sends a user message
|
||||||
|
3. Streams back the response
|
||||||
|
"""
|
||||||
|
|
||||||
|
url = f"ws://{PROXY_URL.replace('http://', '').replace('https://', '')}/v1/realtime?model={MODEL}"
|
||||||
|
headers = {"Authorization": f"Bearer {API_KEY}"}
|
||||||
|
|
||||||
|
print(f"🎙️ Connecting to voice agent...")
|
||||||
|
print(f" Model: {MODEL}")
|
||||||
|
print(f" Proxy: {PROXY_URL}")
|
||||||
|
print()
|
||||||
|
|
||||||
|
async with websockets.connect(url, additional_headers=headers) as ws:
|
||||||
|
# Receive initial connection event
|
||||||
|
initial = json.loads(await ws.recv())
|
||||||
|
print(f"✅ Connected! Event: {initial['type']}\n")
|
||||||
|
|
||||||
|
# Get user input
|
||||||
|
user_message = input("💬 Your message: ").strip()
|
||||||
|
if not user_message:
|
||||||
|
user_message = "Tell me a fun fact about AI!"
|
||||||
|
|
||||||
|
print(f"\n🤖 Sending to {MODEL}...\n")
|
||||||
|
|
||||||
|
# Send user message
|
||||||
|
await ws.send(json.dumps({
|
||||||
|
"type": "conversation.item.create",
|
||||||
|
"item": {
|
||||||
|
"type": "message",
|
||||||
|
"role": "user",
|
||||||
|
"content": [{"type": "input_text", "text": user_message}]
|
||||||
|
}
|
||||||
|
}))
|
||||||
|
|
||||||
|
# Request response
|
||||||
|
await ws.send(json.dumps({
|
||||||
|
"type": "response.create",
|
||||||
|
"response": {"modalities": ["text", "audio"]}
|
||||||
|
}))
|
||||||
|
|
||||||
|
# Stream response
|
||||||
|
print("🎤 Response: ", end='', flush=True)
|
||||||
|
transcript = []
|
||||||
|
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
msg = await asyncio.wait_for(ws.recv(), timeout=15.0)
|
||||||
|
event = json.loads(msg)
|
||||||
|
|
||||||
|
# Capture transcript deltas
|
||||||
|
if event['type'] == 'response.output_audio_transcript.delta':
|
||||||
|
delta = event.get('delta', '')
|
||||||
|
if delta:
|
||||||
|
print(delta, end='', flush=True)
|
||||||
|
transcript.append(delta)
|
||||||
|
|
||||||
|
# Done when response completes
|
||||||
|
elif event['type'] == 'response.done':
|
||||||
|
break
|
||||||
|
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
print("\n")
|
||||||
|
|
||||||
|
if transcript:
|
||||||
|
print(f"✅ Complete response: {''.join(transcript)}")
|
||||||
|
|
||||||
|
await ws.close()
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Run the voice agent"""
|
||||||
|
print("=" * 70)
|
||||||
|
print("LiveKit xAI Voice Agent via LiteLLM Proxy")
|
||||||
|
print("=" * 70)
|
||||||
|
print()
|
||||||
|
|
||||||
|
try:
|
||||||
|
asyncio.run(run_voice_agent())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n\n👋 Goodbye!")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"\n❌ Error: {e}")
|
||||||
|
print("\nMake sure LiteLLM proxy is running:")
|
||||||
|
print(f" litellm --config config.yaml --port 4000")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
2
cookbook/livekit_agent_sdk/requirements.txt
Normal file
2
cookbook/livekit_agent_sdk/requirements.txt
Normal file
|
|
@ -0,0 +1,2 @@
|
||||||
|
livekit-agents[xai]>=1.3.12
|
||||||
|
websockets>=15.0.1
|
||||||
288
cookbook/nova_sonic_realtime.py
Normal file
288
cookbook/nova_sonic_realtime.py
Normal file
|
|
@ -0,0 +1,288 @@
|
||||||
|
"""
|
||||||
|
Client script to test Nova Sonic realtime API through LiteLLM proxy.
|
||||||
|
|
||||||
|
This script connects to LiteLLM proxy's realtime endpoint and enables
|
||||||
|
speech-to-speech conversation with Bedrock Nova Sonic.
|
||||||
|
|
||||||
|
Prerequisites:
|
||||||
|
- LiteLLM proxy running with Bedrock configured
|
||||||
|
- pyaudio installed: pip install pyaudio
|
||||||
|
- websockets installed: pip install websockets
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python nova_sonic_realtime.py
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import pyaudio
|
||||||
|
import websockets
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
# Bounded queue size for audio chunks (configurable via env to avoid unbounded memory)
|
||||||
|
AUDIO_QUEUE_MAXSIZE = int(os.getenv("LITELLM_ASYNCIO_QUEUE_MAXSIZE", 10_000))
|
||||||
|
|
||||||
|
# Audio configuration (matching Nova Sonic requirements)
|
||||||
|
INPUT_SAMPLE_RATE = 16000 # Nova Sonic expects 16kHz input
|
||||||
|
OUTPUT_SAMPLE_RATE = 24000 # Nova Sonic outputs 24kHz
|
||||||
|
CHANNELS = 1
|
||||||
|
FORMAT = pyaudio.paInt16
|
||||||
|
CHUNK_SIZE = 1024
|
||||||
|
|
||||||
|
# LiteLLM proxy configuration
|
||||||
|
LITELLM_PROXY_URL = "ws://localhost:4000/v1/realtime?model=bedrock-sonic"
|
||||||
|
LITELLM_API_KEY = "sk-12345" # Your LiteLLM API key
|
||||||
|
|
||||||
|
|
||||||
|
class RealtimeClient:
|
||||||
|
"""Client for LiteLLM realtime API with audio support."""
|
||||||
|
|
||||||
|
def __init__(self, url: str, api_key: str):
|
||||||
|
self.url = url
|
||||||
|
self.api_key = api_key
|
||||||
|
self.ws: Optional[websockets.WebSocketClientProtocol] = None
|
||||||
|
self.is_active = False
|
||||||
|
self.audio_queue = asyncio.Queue(maxsize=AUDIO_QUEUE_MAXSIZE)
|
||||||
|
self.pyaudio = pyaudio.PyAudio()
|
||||||
|
self.input_stream = None
|
||||||
|
self.output_stream = None
|
||||||
|
|
||||||
|
async def connect(self):
|
||||||
|
"""Connect to LiteLLM proxy realtime endpoint."""
|
||||||
|
print(f"Connecting to {self.url}...")
|
||||||
|
|
||||||
|
headers = {}
|
||||||
|
if self.api_key:
|
||||||
|
headers["Authorization"] = f"Bearer {self.api_key}"
|
||||||
|
|
||||||
|
self.ws = await websockets.connect(
|
||||||
|
self.url,
|
||||||
|
additional_headers=headers,
|
||||||
|
max_size=10 * 1024 * 1024, # 10MB max message size
|
||||||
|
)
|
||||||
|
self.is_active = True
|
||||||
|
print("✓ Connected to LiteLLM proxy")
|
||||||
|
|
||||||
|
async def send_session_update(self):
|
||||||
|
"""Send session configuration."""
|
||||||
|
session_update = {
|
||||||
|
"type": "session.update",
|
||||||
|
"session": {
|
||||||
|
"instructions": "You are a friendly assistant. Keep your responses short and conversational.",
|
||||||
|
"voice": "matthew",
|
||||||
|
"temperature": 0.8,
|
||||||
|
"max_response_output_tokens": 1024,
|
||||||
|
"modalities": ["text", "audio"],
|
||||||
|
"input_audio_format": "pcm16",
|
||||||
|
"output_audio_format": "pcm16",
|
||||||
|
"turn_detection": {
|
||||||
|
"type": "server_vad",
|
||||||
|
"threshold": 0.5,
|
||||||
|
"prefix_padding_ms": 300,
|
||||||
|
"silence_duration_ms": 500,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
await self.ws.send(json.dumps(session_update))
|
||||||
|
print("✓ Session configuration sent")
|
||||||
|
|
||||||
|
async def receive_messages(self):
|
||||||
|
"""Receive and process messages from the server."""
|
||||||
|
try:
|
||||||
|
async for message in self.ws:
|
||||||
|
if not self.is_active:
|
||||||
|
break
|
||||||
|
|
||||||
|
try:
|
||||||
|
data = json.loads(message)
|
||||||
|
event_type = data.get("type")
|
||||||
|
|
||||||
|
if event_type == "session.created":
|
||||||
|
print(f"✓ Session created: {data.get('session', {}).get('id')}")
|
||||||
|
|
||||||
|
elif event_type == "response.created":
|
||||||
|
print("🤖 Assistant is responding...")
|
||||||
|
|
||||||
|
elif event_type == "response.text.delta":
|
||||||
|
# Print text transcription
|
||||||
|
delta = data.get("delta", "")
|
||||||
|
print(delta, end="", flush=True)
|
||||||
|
|
||||||
|
elif event_type == "response.audio.delta":
|
||||||
|
# Queue audio for playback
|
||||||
|
audio_b64 = data.get("delta", "")
|
||||||
|
if audio_b64:
|
||||||
|
audio_bytes = base64.b64decode(audio_b64)
|
||||||
|
await self.audio_queue.put(audio_bytes)
|
||||||
|
|
||||||
|
elif event_type == "response.text.done":
|
||||||
|
print() # New line after text
|
||||||
|
|
||||||
|
elif event_type == "response.done":
|
||||||
|
print("✓ Response complete")
|
||||||
|
|
||||||
|
elif event_type == "error":
|
||||||
|
print(f"❌ Error: {data.get('error', {})}")
|
||||||
|
|
||||||
|
else:
|
||||||
|
# Debug: print other event types
|
||||||
|
print(f"[{event_type}]", end=" ")
|
||||||
|
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
print(f"Failed to parse message: {message[:100]}")
|
||||||
|
|
||||||
|
except websockets.exceptions.ConnectionClosed:
|
||||||
|
print("\n✗ Connection closed")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"\n✗ Error receiving messages: {e}")
|
||||||
|
finally:
|
||||||
|
self.is_active = False
|
||||||
|
|
||||||
|
async def send_audio_chunk(self, audio_bytes: bytes):
|
||||||
|
"""Send audio chunk to server."""
|
||||||
|
if not self.is_active or not self.ws:
|
||||||
|
return
|
||||||
|
|
||||||
|
audio_b64 = base64.b64encode(audio_bytes).decode("utf-8")
|
||||||
|
message = {
|
||||||
|
"type": "input_audio_buffer.append",
|
||||||
|
"audio": audio_b64,
|
||||||
|
}
|
||||||
|
await self.ws.send(json.dumps(message))
|
||||||
|
|
||||||
|
async def commit_audio_buffer(self):
|
||||||
|
"""Commit the audio buffer to trigger processing."""
|
||||||
|
if not self.is_active or not self.ws:
|
||||||
|
return
|
||||||
|
|
||||||
|
message = {"type": "input_audio_buffer.commit"}
|
||||||
|
await self.ws.send(json.dumps(message))
|
||||||
|
|
||||||
|
async def capture_audio(self):
|
||||||
|
"""Capture audio from microphone and send to server."""
|
||||||
|
print("\n🎤 Starting audio capture...")
|
||||||
|
print("Speak into your microphone. Press Ctrl+C to stop.\n")
|
||||||
|
|
||||||
|
self.input_stream = self.pyaudio.open(
|
||||||
|
format=FORMAT,
|
||||||
|
channels=CHANNELS,
|
||||||
|
rate=INPUT_SAMPLE_RATE,
|
||||||
|
input=True,
|
||||||
|
frames_per_buffer=CHUNK_SIZE,
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
while self.is_active:
|
||||||
|
audio_data = self.input_stream.read(CHUNK_SIZE, exception_on_overflow=False)
|
||||||
|
await self.send_audio_chunk(audio_data)
|
||||||
|
await asyncio.sleep(0.01) # Small delay to prevent overwhelming
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error capturing audio: {e}")
|
||||||
|
finally:
|
||||||
|
if self.input_stream:
|
||||||
|
self.input_stream.stop_stream()
|
||||||
|
self.input_stream.close()
|
||||||
|
|
||||||
|
async def play_audio(self):
|
||||||
|
"""Play audio responses from the server."""
|
||||||
|
print("🔊 Starting audio playback...")
|
||||||
|
|
||||||
|
self.output_stream = self.pyaudio.open(
|
||||||
|
format=FORMAT,
|
||||||
|
channels=CHANNELS,
|
||||||
|
rate=OUTPUT_SAMPLE_RATE,
|
||||||
|
output=True,
|
||||||
|
frames_per_buffer=CHUNK_SIZE,
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
while self.is_active:
|
||||||
|
try:
|
||||||
|
audio_data = await asyncio.wait_for(
|
||||||
|
self.audio_queue.get(), timeout=0.1
|
||||||
|
)
|
||||||
|
if audio_data:
|
||||||
|
self.output_stream.write(audio_data)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
continue
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error playing audio: {e}")
|
||||||
|
finally:
|
||||||
|
if self.output_stream:
|
||||||
|
self.output_stream.stop_stream()
|
||||||
|
self.output_stream.close()
|
||||||
|
|
||||||
|
async def close(self):
|
||||||
|
"""Close the connection and cleanup."""
|
||||||
|
self.is_active = False
|
||||||
|
|
||||||
|
if self.ws:
|
||||||
|
await self.ws.close()
|
||||||
|
|
||||||
|
if self.input_stream:
|
||||||
|
self.input_stream.stop_stream()
|
||||||
|
self.input_stream.close()
|
||||||
|
|
||||||
|
if self.output_stream:
|
||||||
|
self.output_stream.stop_stream()
|
||||||
|
self.output_stream.close()
|
||||||
|
|
||||||
|
self.pyaudio.terminate()
|
||||||
|
print("\n✓ Connection closed")
|
||||||
|
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
"""Main function to run the realtime client."""
|
||||||
|
print("=" * 80)
|
||||||
|
print("Bedrock Nova Sonic Realtime Client")
|
||||||
|
print("=" * 80)
|
||||||
|
print()
|
||||||
|
|
||||||
|
client = RealtimeClient(LITELLM_PROXY_URL, LITELLM_API_KEY)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Connect to server
|
||||||
|
await client.connect()
|
||||||
|
|
||||||
|
# Send session configuration
|
||||||
|
await client.send_session_update()
|
||||||
|
|
||||||
|
# Wait a moment for session to be established
|
||||||
|
await asyncio.sleep(0.5)
|
||||||
|
|
||||||
|
# Start tasks
|
||||||
|
receive_task = asyncio.create_task(client.receive_messages())
|
||||||
|
capture_task = asyncio.create_task(client.capture_audio())
|
||||||
|
playback_task = asyncio.create_task(client.play_audio())
|
||||||
|
|
||||||
|
# Wait for user to interrupt
|
||||||
|
await asyncio.gather(
|
||||||
|
receive_task,
|
||||||
|
capture_task,
|
||||||
|
playback_task,
|
||||||
|
return_exceptions=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n\n⚠ Interrupted by user")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"\n❌ Error: {e}")
|
||||||
|
import traceback
|
||||||
|
traceback.print_exc()
|
||||||
|
finally:
|
||||||
|
await client.close()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
print("\nMake sure:")
|
||||||
|
print("1. LiteLLM proxy is running on port 4000")
|
||||||
|
print("2. Bedrock is configured in proxy_server_config.yaml")
|
||||||
|
print("3. AWS credentials are set")
|
||||||
|
print()
|
||||||
|
|
||||||
|
try:
|
||||||
|
asyncio.run(main())
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
print("\n\nGoodbye!")
|
||||||
|
|
@ -8,7 +8,8 @@ WORKDIR /app
|
||||||
COPY config.yaml .
|
COPY config.yaml .
|
||||||
|
|
||||||
# Make sure your docker/entrypoint.sh is executable
|
# Make sure your docker/entrypoint.sh is executable
|
||||||
RUN chmod +x docker/entrypoint.sh
|
# Convert Windows line endings to Unix
|
||||||
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
|
||||||
|
|
||||||
# Expose the necessary port
|
# Expose the necessary port
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
|
||||||
|
|
@ -18,13 +18,17 @@ type: application
|
||||||
# This is the chart version. This version number should be incremented each time you make changes
|
# This is the chart version. This version number should be incremented each time you make changes
|
||||||
# to the chart and its templates, including the app version.
|
# to the chart and its templates, including the app version.
|
||||||
# Versions are expected to follow Semantic Versioning (https://semver.org/)
|
# Versions are expected to follow Semantic Versioning (https://semver.org/)
|
||||||
version: 0.4.10
|
version: 1.1.0
|
||||||
|
|
||||||
# This is the version number of the application being deployed. This version number should be
|
# This is the version number of the application being deployed. This version number should be
|
||||||
# incremented each time you make changes to the application. Versions are not expected to
|
# incremented each time you make changes to the application. Versions are not expected to
|
||||||
# follow Semantic Versioning. They should reflect the version the application is using.
|
# follow Semantic Versioning. They should reflect the version the application is using.
|
||||||
# It is recommended to use it with quotes.
|
# It is recommended to use it with quotes.
|
||||||
appVersion: v1.50.2
|
appVersion: v1.80.12
|
||||||
|
|
||||||
|
annotations:
|
||||||
|
org.opencontainers.image.source: "https://github.com/BerriAI/litellm"
|
||||||
|
org.opencontainers.image.url: "https://docs.litellm.ai/"
|
||||||
|
|
||||||
dependencies:
|
dependencies:
|
||||||
- name: "postgresql"
|
- name: "postgresql"
|
||||||
|
|
|
||||||
|
|
@ -29,7 +29,7 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
||||||
| `masterkey` | The Master API Key for LiteLLM. If not specified, a random key in the `sk-...` format is generated. | N/A |
|
| `masterkey` | The Master API Key for LiteLLM. If not specified, a random key in the `sk-...` format is generated. | N/A |
|
||||||
| `environmentSecrets` | An optional array of Secret object names. The keys and values in these secrets will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
| `environmentSecrets` | An optional array of Secret object names. The keys and values in these secrets will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
||||||
| `environmentConfigMaps` | An optional array of ConfigMap object names. The keys and values in these configmaps will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
| `environmentConfigMaps` | An optional array of ConfigMap object names. The keys and values in these configmaps will be presented to the LiteLLM proxy pod as environment variables. See below for an example Secret object. | `[]` |
|
||||||
| `image.repository` | LiteLLM Proxy image repository | `ghcr.io/berriai/litellm` |
|
| `image.repository` | LiteLLM Proxy image repository | `docker.litellm.ai/berriai/litellm` |
|
||||||
| `image.pullPolicy` | LiteLLM Proxy image pull policy | `IfNotPresent` |
|
| `image.pullPolicy` | LiteLLM Proxy image pull policy | `IfNotPresent` |
|
||||||
| `image.tag` | Overrides the image tag whose default the latest version of LiteLLM at the time this chart was published. | `""` |
|
| `image.tag` | Overrides the image tag whose default the latest version of LiteLLM at the time this chart was published. | `""` |
|
||||||
| `imagePullSecrets` | Registry credentials for the LiteLLM and initContainer images. | `[]` |
|
| `imagePullSecrets` | Registry credentials for the LiteLLM and initContainer images. | `[]` |
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ metadata:
|
||||||
{{- toYaml .Values.deploymentLabels | nindent 4 }}
|
{{- toYaml .Values.deploymentLabels | nindent 4 }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
spec:
|
spec:
|
||||||
{{- if not .Values.autoscaling.enabled }}
|
{{- if and (not .Values.keda.enabled) (not .Values.autoscaling.enabled) }}
|
||||||
replicas: {{ .Values.replicaCount }}
|
replicas: {{ .Values.replicaCount }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
selector:
|
selector:
|
||||||
|
|
@ -38,6 +38,10 @@ spec:
|
||||||
serviceAccountName: {{ include "litellm.serviceAccountName" . }}
|
serviceAccountName: {{ include "litellm.serviceAccountName" . }}
|
||||||
securityContext:
|
securityContext:
|
||||||
{{- toYaml .Values.podSecurityContext | nindent 8 }}
|
{{- toYaml .Values.podSecurityContext | nindent 8 }}
|
||||||
|
{{- with .Values.extraInitContainers }}
|
||||||
|
initContainers:
|
||||||
|
{{- toYaml . | nindent 8 }}
|
||||||
|
{{- end }}
|
||||||
containers:
|
containers:
|
||||||
- name: {{ include "litellm.name" . }}
|
- name: {{ include "litellm.name" . }}
|
||||||
securityContext:
|
securityContext:
|
||||||
|
|
@ -170,7 +174,8 @@ spec:
|
||||||
{{- toYaml .Values.resources | nindent 12 }}
|
{{- toYaml .Values.resources | nindent 12 }}
|
||||||
volumeMounts:
|
volumeMounts:
|
||||||
- name: litellm-config
|
- name: litellm-config
|
||||||
mountPath: /etc/litellm/
|
mountPath: /etc/litellm/config.yaml
|
||||||
|
subPath: config.yaml
|
||||||
{{ if .Values.securityContext.readOnlyRootFilesystem }}
|
{{ if .Values.securityContext.readOnlyRootFilesystem }}
|
||||||
- name: tmp
|
- name: tmp
|
||||||
mountPath: /tmp
|
mountPath: /tmp
|
||||||
|
|
@ -182,6 +187,10 @@ spec:
|
||||||
{{- with .Values.volumeMounts }}
|
{{- with .Values.volumeMounts }}
|
||||||
{{- toYaml . | nindent 12 }}
|
{{- toYaml . | nindent 12 }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
|
{{- with .Values.lifecycle }}
|
||||||
|
lifecycle:
|
||||||
|
{{- toYaml . | nindent 12 }}
|
||||||
|
{{- end }}
|
||||||
{{- with .Values.extraContainers }}
|
{{- with .Values.extraContainers }}
|
||||||
{{- toYaml . | nindent 8 }}
|
{{- toYaml . | nindent 8 }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
|
|
|
||||||
37
deploy/charts/litellm-helm/templates/keda.yaml
Normal file
37
deploy/charts/litellm-helm/templates/keda.yaml
Normal file
|
|
@ -0,0 +1,37 @@
|
||||||
|
{{- if and .Values.keda.enabled (not .Values.autoscaling.enabled) }}
|
||||||
|
apiVersion: keda.sh/v1alpha1
|
||||||
|
kind: ScaledObject
|
||||||
|
metadata:
|
||||||
|
name: {{ include "litellm.fullname" . }}
|
||||||
|
labels:
|
||||||
|
{{- include "litellm.labels" . | nindent 4 }}
|
||||||
|
{{- if .Values.keda.scaledObject.annotations }}
|
||||||
|
annotations: {{ toYaml .Values.keda.scaledObject.annotations | nindent 4 }}
|
||||||
|
{{- end }}
|
||||||
|
spec:
|
||||||
|
scaleTargetRef:
|
||||||
|
name: {{ include "litellm.fullname" . }}
|
||||||
|
pollingInterval: {{ .Values.keda.pollingInterval }}
|
||||||
|
cooldownPeriod: {{ .Values.keda.cooldownPeriod }}
|
||||||
|
minReplicaCount: {{ .Values.keda.minReplicas }}
|
||||||
|
maxReplicaCount: {{ .Values.keda.maxReplicas }}
|
||||||
|
{{- with .Values.keda.fallback }}
|
||||||
|
fallback:
|
||||||
|
failureThreshold: {{ .failureThreshold | default 3 }}
|
||||||
|
replicas: {{ .replicas | default $.Values.keda.maxReplicas }}
|
||||||
|
{{- end }}
|
||||||
|
triggers:
|
||||||
|
{{- with .Values.keda.triggers }}
|
||||||
|
{{- toYaml . | nindent 2 }}
|
||||||
|
{{- end }}
|
||||||
|
advanced:
|
||||||
|
restoreToOriginalReplicaCount: {{ .Values.keda.restoreToOriginalReplicaCount }}
|
||||||
|
{{- if .Values.keda.behavior }}
|
||||||
|
horizontalPodAutoscalerConfig:
|
||||||
|
behavior:
|
||||||
|
{{- with .Values.keda.behavior }}
|
||||||
|
{{- toYaml . | nindent 8 }}
|
||||||
|
{{- end }}
|
||||||
|
|
||||||
|
{{- end }}
|
||||||
|
{{- end }}
|
||||||
|
|
@ -35,6 +35,10 @@ spec:
|
||||||
{{- toYaml . | nindent 8 }}
|
{{- toYaml . | nindent 8 }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
serviceAccountName: {{ include "litellm.serviceAccountName" . }}
|
serviceAccountName: {{ include "litellm.serviceAccountName" . }}
|
||||||
|
{{- with .Values.migrationJob.extraInitContainers }}
|
||||||
|
initContainers:
|
||||||
|
{{- toYaml . | nindent 8 }}
|
||||||
|
{{- end }}
|
||||||
containers:
|
containers:
|
||||||
- name: prisma-migrations
|
- name: prisma-migrations
|
||||||
image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default (printf "main-%s" .Chart.AppVersion) }}"
|
image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default (printf "main-%s" .Chart.AppVersion) }}"
|
||||||
|
|
|
||||||
|
|
@ -136,4 +136,27 @@ tests:
|
||||||
path: spec.template.spec.containers[0].volumeMounts
|
path: spec.template.spec.containers[0].volumeMounts
|
||||||
content:
|
content:
|
||||||
name: litellm-config
|
name: litellm-config
|
||||||
mountPath: /etc/litellm/
|
mountPath: /etc/litellm/config.yaml
|
||||||
|
subPath: config.yaml
|
||||||
|
- it: should work with lifecycle hooks
|
||||||
|
template: deployment.yaml
|
||||||
|
set:
|
||||||
|
lifecycle:
|
||||||
|
preStop:
|
||||||
|
exec:
|
||||||
|
command:
|
||||||
|
- /bin/sh
|
||||||
|
- -c
|
||||||
|
- echo "Container stopping"
|
||||||
|
asserts:
|
||||||
|
- exists:
|
||||||
|
path: spec.template.spec.containers[0].lifecycle
|
||||||
|
- equal:
|
||||||
|
path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[0]
|
||||||
|
value: /bin/sh
|
||||||
|
- equal:
|
||||||
|
path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[1]
|
||||||
|
value: -c
|
||||||
|
- equal:
|
||||||
|
path: spec.template.spec.containers[0].lifecycle.preStop.exec.command[2]
|
||||||
|
value: echo "Container stopping"
|
||||||
|
|
@ -156,6 +156,40 @@ autoscaling:
|
||||||
targetCPUUtilizationPercentage: 80
|
targetCPUUtilizationPercentage: 80
|
||||||
# targetMemoryUtilizationPercentage: 80
|
# targetMemoryUtilizationPercentage: 80
|
||||||
|
|
||||||
|
# Autoscaling with keda is mutually exclusive with hpa
|
||||||
|
keda:
|
||||||
|
enabled: false
|
||||||
|
minReplicas: 1
|
||||||
|
maxReplicas: 100
|
||||||
|
pollingInterval: 30
|
||||||
|
cooldownPeriod: 300
|
||||||
|
# fallback:
|
||||||
|
# failureThreshold: 3
|
||||||
|
# replicas: 11
|
||||||
|
restoreToOriginalReplicaCount: false
|
||||||
|
scaledObject:
|
||||||
|
annotations: {}
|
||||||
|
triggers: []
|
||||||
|
# - type: prometheus
|
||||||
|
# metadata:
|
||||||
|
# serverAddress: http://<prometheus-host>:9090
|
||||||
|
# metricName: http_requests_total
|
||||||
|
# threshold: '100'
|
||||||
|
# query: sum(rate(http_requests_total{deployment="my-deployment"}[2m]))
|
||||||
|
behavior: {}
|
||||||
|
# scaleDown:
|
||||||
|
# stabilizationWindowSeconds: 300
|
||||||
|
# policies:
|
||||||
|
# - type: Pods
|
||||||
|
# value: 1
|
||||||
|
# periodSeconds: 180
|
||||||
|
# scaleUp:
|
||||||
|
# stabilizationWindowSeconds: 300
|
||||||
|
# policies:
|
||||||
|
# - type: Pods
|
||||||
|
# value: 2
|
||||||
|
# periodSeconds: 60
|
||||||
|
|
||||||
# Additional volumes on the output Deployment definition.
|
# Additional volumes on the output Deployment definition.
|
||||||
volumes: []
|
volumes: []
|
||||||
# - name: foo
|
# - name: foo
|
||||||
|
|
@ -200,6 +234,14 @@ db:
|
||||||
# instance. See the "postgresql" top level key for additional configuration.
|
# instance. See the "postgresql" top level key for additional configuration.
|
||||||
deployStandalone: true
|
deployStandalone: true
|
||||||
|
|
||||||
|
# Lifecycle hooks for the LiteLLM container
|
||||||
|
# Example:
|
||||||
|
# lifecycle:
|
||||||
|
# preStop:
|
||||||
|
# exec:
|
||||||
|
# command: ["/bin/sh", "-c", "sleep 10"]
|
||||||
|
lifecycle: {}
|
||||||
|
|
||||||
# Settings for Bitnami postgresql chart (if db.deployStandalone is true, ignored
|
# Settings for Bitnami postgresql chart (if db.deployStandalone is true, ignored
|
||||||
# otherwise)
|
# otherwise)
|
||||||
postgresql:
|
postgresql:
|
||||||
|
|
@ -239,6 +281,7 @@ migrationJob:
|
||||||
# cpu: 100m
|
# cpu: 100m
|
||||||
# memory: 100Mi
|
# memory: 100Mi
|
||||||
extraContainers: []
|
extraContainers: []
|
||||||
|
extraInitContainers: []
|
||||||
|
|
||||||
# Hook configuration
|
# Hook configuration
|
||||||
hooks:
|
hooks:
|
||||||
|
|
|
||||||
46
docker-compose.hardened.yml
Normal file
46
docker-compose.hardened.yml
Normal file
|
|
@ -0,0 +1,46 @@
|
||||||
|
services:
|
||||||
|
# Hardened stack: for testing the proxy under non-root, read-only, proxy-enforced constraints.
|
||||||
|
# Keep this file focused on hardening/QA scenarios; leave the main docker-compose.yml for default dev usage.
|
||||||
|
litellm:
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: docker/Dockerfile.non_root
|
||||||
|
target: runtime
|
||||||
|
args:
|
||||||
|
PROXY_EXTRAS_SOURCE: "local"
|
||||||
|
depends_on:
|
||||||
|
- squid
|
||||||
|
user: "101:101"
|
||||||
|
group_add:
|
||||||
|
- "2345"
|
||||||
|
read_only: true
|
||||||
|
cap_drop:
|
||||||
|
- ALL
|
||||||
|
security_opt:
|
||||||
|
- no-new-privileges:true
|
||||||
|
tmpfs:
|
||||||
|
- /app/cache:rw,noexec,nosuid,nodev,size=128m,uid=101,gid=101,mode=1777
|
||||||
|
- /app/migrations:rw,noexec,nosuid,nodev,size=64m,uid=101,gid=101,mode=1777
|
||||||
|
volumes:
|
||||||
|
- ./proxy_server_config.yaml:/app/config.yaml:ro
|
||||||
|
environment:
|
||||||
|
LITELLM_NON_ROOT: "true"
|
||||||
|
PRISMA_BINARY_CACHE_DIR: "/app/cache/prisma-python/binaries"
|
||||||
|
XDG_CACHE_HOME: "/app/cache"
|
||||||
|
LITELLM_MIGRATION_DIR: "/app/migrations"
|
||||||
|
HTTP_PROXY: "http://squid:3128"
|
||||||
|
HTTPS_PROXY: "http://squid:3128"
|
||||||
|
NO_PROXY: "localhost,127.0.0.1,db"
|
||||||
|
command:
|
||||||
|
- "--port"
|
||||||
|
- "4000"
|
||||||
|
- "--config"
|
||||||
|
- "/app/config.yaml"
|
||||||
|
squid:
|
||||||
|
image: sameersbn/squid:3.5.27-2
|
||||||
|
restart: unless-stopped
|
||||||
|
ports:
|
||||||
|
- "3128:3128"
|
||||||
|
tmpfs:
|
||||||
|
- /var/spool/squid:rw,noexec,nosuid,nodev,size=64m
|
||||||
|
- /var/log/squid:rw,noexec,nosuid,nodev,size=16m
|
||||||
|
|
@ -4,7 +4,7 @@ services:
|
||||||
context: .
|
context: .
|
||||||
args:
|
args:
|
||||||
target: runtime
|
target: runtime
|
||||||
image: ghcr.io/berriai/litellm:main-stable
|
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||||
#########################################
|
#########################################
|
||||||
## Uncomment these lines to start proxy with a config.yaml file ##
|
## Uncomment these lines to start proxy with a config.yaml file ##
|
||||||
# volumes:
|
# volumes:
|
||||||
|
|
|
||||||
|
|
@ -34,8 +34,8 @@ RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
||||||
# Runtime stage
|
# Runtime stage
|
||||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||||
|
|
||||||
# Update dependencies and clean up
|
# Update dependencies and clean up, install libsndfile for audio processing
|
||||||
RUN apk upgrade --no-cache
|
RUN apk upgrade --no-cache && apk add --no-cache libsndfile
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
|
|
@ -46,8 +46,9 @@ COPY --from=builder /wheels/ /wheels/
|
||||||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||||
|
|
||||||
RUN chmod +x docker/entrypoint.sh
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
RUN chmod +x docker/prod_entrypoint.sh
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
|
||||||
|
RUN sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||||
|
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,19 @@ FROM ghcr.io/berriai/litellm:litellm_fwd_server_root_path-dev
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# Install Node.js and npm (adjust version as needed)
|
# Install Node.js and npm (adjust version as needed)
|
||||||
RUN apt-get update && apt-get install -y nodejs npm
|
RUN apt-get update && apt-get install -y nodejs npm && \
|
||||||
|
npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \
|
||||||
|
GLOBAL="$(npm root -g)" && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done && \
|
||||||
|
npm cache clean --force
|
||||||
|
|
||||||
# Copy the UI source into the container
|
# Copy the UI source into the container
|
||||||
COPY ./ui/litellm-dashboard /app/ui/litellm-dashboard
|
COPY ./ui/litellm-dashboard /app/ui/litellm-dashboard
|
||||||
|
|
@ -32,8 +44,9 @@ RUN rm -rf /app/litellm/proxy/_experimental/out/* && \
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# Make sure your docker/entrypoint.sh is executable
|
# Make sure your docker/entrypoint.sh is executable
|
||||||
RUN chmod +x docker/entrypoint.sh
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
RUN chmod +x docker/prod_entrypoint.sh
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
|
||||||
|
RUN sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||||
|
|
||||||
# Expose the necessary port
|
# Expose the necessary port
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
|
||||||
|
|
@ -27,7 +27,8 @@ RUN python -m pip install build
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# Build Admin UI
|
# Build Admin UI
|
||||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
# Convert Windows line endings to Unix and make executable
|
||||||
|
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||||
|
|
||||||
# Build the package
|
# Build the package
|
||||||
RUN rm -rf dist/* && python -m build
|
RUN rm -rf dist/* && python -m build
|
||||||
|
|
@ -48,7 +49,19 @@ FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||||
USER root
|
USER root
|
||||||
|
|
||||||
# Install runtime dependencies
|
# Install runtime dependencies
|
||||||
RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip
|
RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip libsndfile && \
|
||||||
|
npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 && \
|
||||||
|
GLOBAL="$(npm root -g)" && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done && \
|
||||||
|
npm cache clean --force
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
# Copy the current directory contents into the container at /app
|
# Copy the current directory contents into the container at /app
|
||||||
|
|
@ -62,21 +75,38 @@ COPY --from=builder /wheels/ /wheels/
|
||||||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||||
|
|
||||||
|
# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete
|
||||||
|
# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/.
|
||||||
|
# Patch every copy of tar, glob, and brace-expansion inside that tree.
|
||||||
|
RUN GLOBAL="$(npm root -g)" && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done
|
||||||
|
|
||||||
# Install semantic_router and aurelio-sdk using script
|
# Install semantic_router and aurelio-sdk using script
|
||||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
# Convert Windows line endings to Unix and make executable
|
||||||
|
RUN sed -i 's/\r$//' docker/install_auto_router.sh && chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
||||||
|
|
||||||
# ensure pyjwt is used, not jwt
|
# ensure pyjwt is used, not jwt
|
||||||
RUN pip uninstall jwt -y
|
RUN pip uninstall jwt -y
|
||||||
RUN pip uninstall PyJWT -y
|
RUN pip uninstall PyJWT -y
|
||||||
RUN pip install PyJWT==2.9.0 --no-cache-dir
|
RUN pip install PyJWT==2.9.0 --no-cache-dir
|
||||||
|
|
||||||
# Build Admin UI
|
# Build Admin UI (runtime stage)
|
||||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
# Convert Windows line endings to Unix and make executable
|
||||||
|
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||||
|
|
||||||
# Generate prisma client
|
# Generate prisma client
|
||||||
RUN prisma generate
|
RUN prisma generate
|
||||||
RUN chmod +x docker/entrypoint.sh
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
RUN chmod +x docker/prod_entrypoint.sh
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
|
||||||
|
RUN sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
||||||
RUN apk add --no-cache supervisor
|
RUN apk add --no-cache supervisor
|
||||||
|
|
|
||||||
|
|
@ -40,7 +40,8 @@ COPY enterprise/ ./enterprise/
|
||||||
COPY docker/ ./docker/
|
COPY docker/ ./docker/
|
||||||
|
|
||||||
# Build Admin UI once
|
# Build Admin UI once
|
||||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
# Convert Windows line endings to Unix and make executable
|
||||||
|
RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||||
|
|
||||||
# Build the package
|
# Build the package
|
||||||
RUN rm -rf dist/* && python -m build
|
RUN rm -rf dist/* && python -m build
|
||||||
|
|
@ -60,7 +61,19 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
libatomic1 \
|
libatomic1 \
|
||||||
nodejs \
|
nodejs \
|
||||||
npm \
|
npm \
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
&& rm -rf /var/lib/apt/lists/* \
|
||||||
|
&& npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 \
|
||||||
|
&& GLOBAL="$(npm root -g)" \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done \
|
||||||
|
&& npm cache clean --force
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
|
|
@ -78,9 +91,27 @@ RUN pip install --no-cache-dir *.whl /wheels/* --no-index --find-links=/wheels/
|
||||||
rm -f *.whl && \
|
rm -f *.whl && \
|
||||||
rm -rf /wheels
|
rm -rf /wheels
|
||||||
|
|
||||||
|
# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete
|
||||||
|
# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/.
|
||||||
|
# Patch every copy of tar, glob, and brace-expansion inside that tree.
|
||||||
|
RUN GLOBAL="$(npm root -g)" && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done
|
||||||
|
|
||||||
# Generate prisma client and set permissions
|
# Generate prisma client and set permissions
|
||||||
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
RUN prisma generate && \
|
RUN prisma generate && \
|
||||||
chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh
|
sed -i 's/\r$//' docker/entrypoint.sh && \
|
||||||
|
sed -i 's/\r$//' docker/prod_entrypoint.sh && \
|
||||||
|
chmod +x docker/entrypoint.sh && \
|
||||||
|
chmod +x docker/prod_entrypoint.sh
|
||||||
|
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
||||||
|
|
|
||||||
16
docker/Dockerfile.health_check
Normal file
16
docker/Dockerfile.health_check
Normal file
|
|
@ -0,0 +1,16 @@
|
||||||
|
FROM python:3.11-slim
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
# Copy health check script and requirements
|
||||||
|
COPY scripts/health_check/health_check_client.py /app/health_check_client.py
|
||||||
|
COPY scripts/health_check/health_check_requirements.txt /app/requirements.txt
|
||||||
|
|
||||||
|
# Install dependencies
|
||||||
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
# Make script executable
|
||||||
|
RUN chmod +x /app/health_check_client.py
|
||||||
|
|
||||||
|
# Set entrypoint
|
||||||
|
ENTRYPOINT ["python", "/app/health_check_client.py"]
|
||||||
|
|
@ -1,154 +1,217 @@
|
||||||
# Base images
|
# Base images
|
||||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
|
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
|
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
|
||||||
|
ARG PROXY_EXTRAS_SOURCE=published
|
||||||
|
|
||||||
# -----------------
|
# -----------------
|
||||||
# Builder Stage
|
# Builder Stage
|
||||||
# -----------------
|
# -----------------
|
||||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||||
|
ARG PROXY_EXTRAS_SOURCE
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# Install build dependencies including Node.js for UI build
|
|
||||||
USER root
|
USER root
|
||||||
|
|
||||||
|
# Install build dependencies with retry logic (includes node for UI build)
|
||||||
RUN for i in 1 2 3; do \
|
RUN for i in 1 2 3; do \
|
||||||
apk add --no-cache \
|
apk add --no-cache \
|
||||||
python3 \
|
python3 \
|
||||||
py3-pip \
|
python3-dev \
|
||||||
clang \
|
py3-pip \
|
||||||
llvm \
|
clang \
|
||||||
lld \
|
llvm \
|
||||||
gcc \
|
lld \
|
||||||
linux-headers \
|
gcc \
|
||||||
build-base \
|
linux-headers \
|
||||||
bash \
|
build-base \
|
||||||
nodejs \
|
bash \
|
||||||
npm && break || sleep 5; \
|
nodejs \
|
||||||
done \
|
npm && break || sleep 5; \
|
||||||
|
done \
|
||||||
&& pip install --no-cache-dir --upgrade pip build
|
&& pip install --no-cache-dir --upgrade pip build
|
||||||
|
|
||||||
# Copy project files
|
# Cache Python dependencies
|
||||||
|
COPY requirements.txt .
|
||||||
|
RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt \
|
||||||
|
&& pip wheel --no-cache-dir --wheel-dir=/wheels/ "semantic_router==0.1.11" "aurelio-sdk==0.0.19" "PyJWT==2.9.0"
|
||||||
|
|
||||||
|
# Copy source after dependency layers
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# Set LITELLM_NON_ROOT flag for build time
|
# Set non-root flag for build time consistency
|
||||||
ENV LITELLM_NON_ROOT=true
|
ENV LITELLM_NON_ROOT=true
|
||||||
|
|
||||||
# Build Admin UI
|
# Build Admin UI using the upstream command order while keeping a single RUN layer
|
||||||
RUN mkdir -p /tmp/litellm_ui
|
RUN mkdir -p /var/lib/litellm/ui && \
|
||||||
|
npm install -g npm@latest && npm cache clean --force && \
|
||||||
|
cd /app/ui/litellm-dashboard && \
|
||||||
|
if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
|
||||||
|
cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
|
||||||
|
fi && \
|
||||||
|
npm install --legacy-peer-deps && \
|
||||||
|
npm run build && \
|
||||||
|
cp -r /app/ui/litellm-dashboard/out/* /var/lib/litellm/ui/ && \
|
||||||
|
mkdir -p /var/lib/litellm/assets && \
|
||||||
|
cp /app/litellm/proxy/logo.jpg /var/lib/litellm/assets/logo.jpg && \
|
||||||
|
( cd /var/lib/litellm/ui && \
|
||||||
|
for html_file in *.html; do \
|
||||||
|
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
|
||||||
|
folder_name="${html_file%.html}" && \
|
||||||
|
mkdir -p "$folder_name" && \
|
||||||
|
mv "$html_file" "$folder_name/index.html"; \
|
||||||
|
fi; \
|
||||||
|
done && \
|
||||||
|
touch .litellm_ui_ready ) && \
|
||||||
|
cd /app/ui/litellm-dashboard && rm -rf ./out
|
||||||
|
|
||||||
RUN npm install -g npm@latest && npm cache clean --force
|
# Build litellm wheel and place it in wheels dir (replace any PyPI wheels)
|
||||||
|
|
||||||
RUN cd /app/ui/litellm-dashboard && \
|
|
||||||
if [ -f "/app/enterprise/enterprise_ui/enterprise_colors.json" ]; then \
|
|
||||||
cp /app/enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
RUN cd /app/ui/litellm-dashboard && rm -f package-lock.json
|
|
||||||
|
|
||||||
RUN cd /app/ui/litellm-dashboard && npm install --legacy-peer-deps
|
|
||||||
|
|
||||||
RUN cd /app/ui/litellm-dashboard && npm run build
|
|
||||||
|
|
||||||
RUN cp -r /app/ui/litellm-dashboard/out/* /tmp/litellm_ui/
|
|
||||||
RUN mkdir -p /tmp/litellm_assets && cp /app/litellm/proxy/logo.jpg /tmp/litellm_assets/logo.jpg
|
|
||||||
|
|
||||||
RUN cd /tmp/litellm_ui && \
|
|
||||||
for html_file in *.html; do \
|
|
||||||
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
|
|
||||||
folder_name="${html_file%.html}" && \
|
|
||||||
mkdir -p "$folder_name" && \
|
|
||||||
mv "$html_file" "$folder_name/index.html"; \
|
|
||||||
fi; \
|
|
||||||
done
|
|
||||||
|
|
||||||
RUN cd /app/ui/litellm-dashboard && rm -rf ./out
|
|
||||||
|
|
||||||
# Build package and wheel dependencies
|
|
||||||
RUN rm -rf dist/* && python -m build && \
|
RUN rm -rf dist/* && python -m build && \
|
||||||
pip install dist/*.whl && \
|
rm -f /wheels/litellm-*.whl && \
|
||||||
pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
cp dist/*.whl /wheels/
|
||||||
|
|
||||||
|
# Optionally build local litellm-proxy-extras wheel
|
||||||
|
RUN if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
|
||||||
|
cd /app/litellm-proxy-extras && rm -rf dist && python -m build && \
|
||||||
|
cp dist/*.whl /wheels/; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Pre-cache Prisma binaries in the builder stage
|
||||||
|
ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
|
||||||
|
PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
|
||||||
|
XDG_CACHE_HOME=/app/.cache \
|
||||||
|
PATH="/usr/lib/python3.13/site-packages/nodejs/bin:${PATH}"
|
||||||
|
|
||||||
|
RUN pip install --no-cache-dir prisma==0.11.0 nodejs-wheel-binaries==24.12.0 \
|
||||||
|
&& mkdir -p /app/.cache/npm
|
||||||
|
|
||||||
|
RUN NPM_CONFIG_CACHE=/app/.cache/npm \
|
||||||
|
python -c "import prisma.cli.prisma as p; p.ensure_cached()"
|
||||||
|
|
||||||
|
RUN prisma generate && \
|
||||||
|
prisma --version && \
|
||||||
|
prisma migrate diff --from-empty --to-schema-datamodel ./schema.prisma --script > /dev/null 2>&1 || true
|
||||||
|
|
||||||
# -----------------
|
# -----------------
|
||||||
# Runtime Stage
|
# Runtime Stage
|
||||||
# -----------------
|
# -----------------
|
||||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||||
|
ARG PROXY_EXTRAS_SOURCE
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# Install runtime dependencies
|
|
||||||
USER root
|
USER root
|
||||||
RUN for i in 1 2 3; do \
|
|
||||||
apk upgrade --no-cache && break || sleep 5; \
|
|
||||||
done \
|
|
||||||
&& for i in 1 2 3; do \
|
|
||||||
apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
|
|
||||||
done
|
|
||||||
|
|
||||||
# Copy only necessary artifacts from builder stage for runtime
|
# Install runtime dependencies with retry
|
||||||
COPY . .
|
RUN for i in 1 2 3; do \
|
||||||
|
apk upgrade --no-cache && break || sleep 5; \
|
||||||
|
done \
|
||||||
|
&& for i in 1 2 3; do \
|
||||||
|
apk add --no-cache python3 py3-pip bash openssl tzdata nodejs npm supervisor && break || sleep 5; \
|
||||||
|
done \
|
||||||
|
&& npm install -g npm@latest tar@7.5.7 glob@11.1.0 @isaacs/brace-expansion@5.0.1 \
|
||||||
|
&& GLOBAL="$(npm root -g)" \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "tar" -path "*/node_modules/tar" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "glob" -path "*/node_modules/glob" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done \
|
||||||
|
&& find "$GLOBAL/npm" -type d -name "brace-expansion" -path "*/node_modules/@isaacs/brace-expansion" | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done \
|
||||||
|
&& npm cache clean --force
|
||||||
|
|
||||||
|
# Copy artifacts from builder
|
||||||
|
COPY --from=builder /app/requirements.txt /app/requirements.txt
|
||||||
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
|
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
|
||||||
COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
|
COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
|
||||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
COPY --from=builder /app/schema.prisma /app/
|
||||||
COPY --from=builder /app/dist/*.whl .
|
# Copy prisma_migration.py for Helm migrations job compatibility
|
||||||
|
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py
|
||||||
COPY --from=builder /wheels/ /wheels/
|
COPY --from=builder /wheels/ /wheels/
|
||||||
COPY --from=builder /tmp/litellm_ui /tmp/litellm_ui
|
COPY --from=builder /var/lib/litellm/ui /var/lib/litellm/ui
|
||||||
COPY --from=builder /tmp/litellm_assets /tmp/litellm_assets
|
COPY --from=builder /var/lib/litellm/assets /var/lib/litellm/assets
|
||||||
|
COPY --from=builder /app/.cache /app/.cache
|
||||||
|
COPY --from=builder /app/litellm-proxy-extras /app/litellm-proxy-extras
|
||||||
|
COPY --from=builder \
|
||||||
|
/usr/lib/python3.13/site-packages/nodejs* \
|
||||||
|
/usr/lib/python3.13/site-packages/prisma* \
|
||||||
|
/usr/lib/python3.13/site-packages/tomlkit* \
|
||||||
|
/usr/lib/python3.13/site-packages/nodeenv* \
|
||||||
|
/usr/lib/python3.13/site-packages/
|
||||||
|
COPY --from=builder /usr/bin/prisma /usr/bin/prisma
|
||||||
|
|
||||||
# Install package from wheel and dependencies
|
# Final runtime environment configuration
|
||||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
|
ENV PRISMA_BINARY_CACHE_DIR=/app/.cache/prisma-python/binaries \
|
||||||
&& rm -f *.whl \
|
PRISMA_CLI_BINARY_TARGETS="debian-openssl-3.0.x" \
|
||||||
&& rm -rf /wheels
|
HOME=/app \
|
||||||
|
LITELLM_NON_ROOT=true \
|
||||||
|
XDG_CACHE_HOME=/app/.cache
|
||||||
|
|
||||||
# Remove test files and keys from dependencies
|
# Install packages from wheels and optional extras without network
|
||||||
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
RUN pip install --no-index --find-links=/wheels/ -r requirements.txt && \
|
||||||
find /usr/lib -type d -path "*/tornado/test" -delete
|
pip install --no-index --find-links=/wheels/ /wheels/litellm-*-py3-none-any.whl && \
|
||||||
|
pip install --no-index --find-links=/wheels/ --no-deps semantic_router==0.1.11 && \
|
||||||
|
pip install --no-index --find-links=/wheels/ aurelio-sdk==0.0.19 && \
|
||||||
|
if [ "$PROXY_EXTRAS_SOURCE" = "local" ]; then \
|
||||||
|
if ls /wheels/litellm_proxy_extras-*.whl >/dev/null 2>&1; then \
|
||||||
|
pip install --no-index --find-links=/wheels/ /wheels/litellm_proxy_extras-*.whl; \
|
||||||
|
else \
|
||||||
|
echo "litellm_proxy_extras wheel not found; skipping local install"; \
|
||||||
|
fi; \
|
||||||
|
fi
|
||||||
|
|
||||||
# Install semantic_router and aurelio-sdk using script
|
# SECURITY FIX: nodejs-wheel-binaries (pip package used by Prisma) bundles a complete
|
||||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
# npm with old vulnerable deps at /usr/lib/python3.*/site-packages/nodejs_wheel/.
|
||||||
|
# Patch every copy of tar, glob, and brace-expansion inside that tree.
|
||||||
|
RUN GLOBAL="$(npm root -g)" && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/tar" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/tar" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/glob" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/glob" "$d"; \
|
||||||
|
done && \
|
||||||
|
find /usr/lib -path "*/nodejs_wheel/*/node_modules/@isaacs/brace-expansion" -type d | while read d; do \
|
||||||
|
rm -rf "$d" && cp -rL "$GLOBAL/@isaacs/brace-expansion" "$d"; \
|
||||||
|
done
|
||||||
|
|
||||||
# Ensure correct JWT library is used (pyjwt not jwt)
|
# Permissions, cleanup, and Prisma prep
|
||||||
RUN pip uninstall jwt -y && \
|
# Convert Windows line endings to Unix for entrypoint scripts
|
||||||
pip uninstall PyJWT -y && \
|
RUN sed -i 's/\r$//' docker/entrypoint.sh && \
|
||||||
pip install PyJWT==2.9.0 --no-cache-dir
|
sed -i 's/\r$//' docker/prod_entrypoint.sh && \
|
||||||
|
chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh && \
|
||||||
|
mkdir -p /nonexistent /.npm /var/lib/litellm/assets /var/lib/litellm/ui && \
|
||||||
|
chown -R nobody:nogroup /app /var/lib/litellm/ui /var/lib/litellm/assets /nonexistent /.npm && \
|
||||||
|
pip uninstall jwt -y || true && \
|
||||||
|
pip uninstall PyJWT -y || true && \
|
||||||
|
pip install --no-index --find-links=/wheels/ PyJWT==2.10.1 --no-cache-dir && \
|
||||||
|
rm -rf /wheels && \
|
||||||
|
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||||
|
chown -R nobody:nogroup $PRISMA_PATH && \
|
||||||
|
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
||||||
|
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH && \
|
||||||
|
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
||||||
|
chgrp -R 0 $PRISMA_PATH /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||||
|
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||||
|
chmod -R g=u $PRISMA_PATH /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||||
|
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||||
|
chmod -R g+w $PRISMA_PATH /var/lib/litellm/ui /var/lib/litellm/assets && \
|
||||||
|
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||||
|
chmod -R g+rX $PRISMA_PATH && \
|
||||||
|
chmod -R g+rX /app/.cache && \
|
||||||
|
mkdir -p /tmp/.npm /nonexistent /.npm
|
||||||
|
|
||||||
# Set Prisma cache directories
|
# Switch to non-root user for runtime
|
||||||
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
|
|
||||||
ENV NPM_CONFIG_CACHE=/.npm
|
|
||||||
|
|
||||||
# Install prisma and make entrypoints executable
|
|
||||||
RUN pip install --no-cache-dir prisma && \
|
|
||||||
chmod +x docker/entrypoint.sh && \
|
|
||||||
chmod +x docker/prod_entrypoint.sh
|
|
||||||
|
|
||||||
# Create directories and set permissions for non-root user
|
|
||||||
RUN mkdir -p /nonexistent /.npm /tmp/litellm_assets && \
|
|
||||||
chown -R nobody:nogroup /app /tmp/litellm_ui /tmp/litellm_assets /nonexistent /.npm && \
|
|
||||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
|
||||||
chown -R nobody:nogroup $PRISMA_PATH && \
|
|
||||||
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
|
||||||
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH
|
|
||||||
|
|
||||||
# OpenShift compatibility
|
|
||||||
RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
|
||||||
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
|
||||||
chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
|
||||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
|
|
||||||
chmod -R g=u $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
|
||||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
|
|
||||||
chmod -R g+w $PRISMA_PATH /tmp/litellm_ui /tmp/litellm_assets && \
|
|
||||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true
|
|
||||||
|
|
||||||
# Switch to non-root user
|
|
||||||
USER nobody
|
USER nobody
|
||||||
|
|
||||||
# Set HOME for prisma generate to have a writable directory
|
# Generate Prisma client as nobody user to ensure correct file ownership
|
||||||
ENV HOME=/app
|
|
||||||
|
|
||||||
# Set LITELLM_NON_ROOT flag for runtime
|
|
||||||
ENV LITELLM_NON_ROOT=true
|
|
||||||
|
|
||||||
RUN prisma generate
|
RUN prisma generate
|
||||||
|
|
||||||
|
# Prisma runtime knobs for offline containers
|
||||||
|
ENV PRISMA_SKIP_POSTINSTALL_GENERATE=1 \
|
||||||
|
PRISMA_HIDE_UPDATE_MESSAGE=1 \
|
||||||
|
PRISMA_ENGINES_CHECKSUM_IGNORE_MISSING=1 \
|
||||||
|
NPM_CONFIG_CACHE=/app/.cache/npm \
|
||||||
|
NPM_CONFIG_PREFER_OFFLINE=true \
|
||||||
|
PRISMA_OFFLINE_MODE=true
|
||||||
|
|
||||||
EXPOSE 4000/tcp
|
EXPOSE 4000/tcp
|
||||||
|
|
||||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||||
|
CMD ["--port", "4000"]
|
||||||
CMD ["--port", "4000"]
|
|
||||||
|
|
|
||||||
|
|
@ -59,6 +59,33 @@ To stop the running containers, use the following command:
|
||||||
docker compose down
|
docker compose down
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Hardened / Offline Testing
|
||||||
|
|
||||||
|
To ensure changes are safe for non-root, read-only root filesystems and restricted egress, always validate with the hardened compose file:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker compose -f docker-compose.yml -f docker-compose.hardened.yml build --no-cache
|
||||||
|
docker compose -f docker-compose.yml -f docker-compose.hardened.yml up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
This setup:
|
||||||
|
- Builds from `docker/Dockerfile.non_root` with Prisma engines and Node toolchain baked into the image.
|
||||||
|
- Runs the proxy as a non-root user with a read-only rootfs and only writable tmpfs mounts:
|
||||||
|
- `/app/cache` (Prisma/NPM cache; backing `PRISMA_BINARY_CACHE_DIR`, `NPM_CONFIG_CACHE`, `XDG_CACHE_HOME`)
|
||||||
|
- `/app/migrations` (Prisma migration workspace; backing `LITELLM_MIGRATION_DIR`)
|
||||||
|
- Pre-builds and serves the admin UI from read-only paths:
|
||||||
|
- `/var/lib/litellm/ui` (pre-restructured Next.js UI with `.litellm_ui_ready` marker)
|
||||||
|
- `/var/lib/litellm/assets` (UI logos and assets)
|
||||||
|
- Routes all outbound traffic through a local Squid proxy that denies egress, so Prisma migrations must use the cached CLI and engines.
|
||||||
|
|
||||||
|
You should also verify offline Prisma behaviour with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --rm --network none --entrypoint prisma ghcr.io/berriai/litellm:main-stable --version
|
||||||
|
```
|
||||||
|
|
||||||
|
This command should succeed (showing engine versions) even with `--network none`, confirming that Prisma binaries are available without network access.
|
||||||
|
|
||||||
## Troubleshooting
|
## Troubleshooting
|
||||||
|
|
||||||
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
|
- **`build_admin_ui.sh: not found`**: This error can occur if the Docker build context is not set correctly. Ensure that you are running the `docker-compose` command from the root of the project.
|
||||||
|
|
|
||||||
|
|
@ -2,6 +2,7 @@
|
||||||
|
|
||||||
if [ "$SEPARATE_HEALTH_APP" = "1" ]; then
|
if [ "$SEPARATE_HEALTH_APP" = "1" ]; then
|
||||||
export LITELLM_ARGS="$@"
|
export LITELLM_ARGS="$@"
|
||||||
|
export SUPERVISORD_STOPWAITSECS="${SUPERVISORD_STOPWAITSECS:-3600}"
|
||||||
exec supervisord -c /etc/supervisord.conf
|
exec supervisord -c /etc/supervisord.conf
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,8 @@
|
||||||
[supervisord]
|
[supervisord]
|
||||||
nodaemon=true
|
nodaemon=true
|
||||||
loglevel=info
|
loglevel=info
|
||||||
|
logfile=/tmp/supervisord.log
|
||||||
|
pidfile=/tmp/supervisord.pid
|
||||||
|
|
||||||
[group:litellm]
|
[group:litellm]
|
||||||
programs=main,health
|
programs=main,health
|
||||||
|
|
@ -14,6 +16,7 @@ priority=1
|
||||||
exitcodes=0
|
exitcodes=0
|
||||||
stopasgroup=true
|
stopasgroup=true
|
||||||
killasgroup=true
|
killasgroup=true
|
||||||
|
stopwaitsecs=%(ENV_SUPERVISORD_STOPWAITSECS)s
|
||||||
stdout_logfile=/dev/stdout
|
stdout_logfile=/dev/stdout
|
||||||
stderr_logfile=/dev/stderr
|
stderr_logfile=/dev/stderr
|
||||||
stdout_logfile_maxbytes = 0
|
stdout_logfile_maxbytes = 0
|
||||||
|
|
@ -29,6 +32,7 @@ priority=2
|
||||||
exitcodes=0
|
exitcodes=0
|
||||||
stopasgroup=true
|
stopasgroup=true
|
||||||
killasgroup=true
|
killasgroup=true
|
||||||
|
stopwaitsecs=%(ENV_SUPERVISORD_STOPWAITSECS)s
|
||||||
stdout_logfile=/dev/stdout
|
stdout_logfile=/dev/stdout
|
||||||
stderr_logfile=/dev/stderr
|
stderr_logfile=/dev/stderr
|
||||||
stdout_logfile_maxbytes = 0
|
stdout_logfile_maxbytes = 0
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,7 @@ authors:
|
||||||
- name: Sameer Kankute
|
- name: Sameer Kankute
|
||||||
title: SWE @ LiteLLM (LLM Translation)
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
url: https://www.linkedin.com/in/sameer-kankute/
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQHB_loQYd5gjg/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1719137160975?e=1765411200&v=beta&t=c8396f--_lH6Fb_pVvx_jGholPfcl0bvwmNynbNdnII
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
- name: Krrish Dholakia
|
- name: Krrish Dholakia
|
||||||
title: "CEO, LiteLLM"
|
title: "CEO, LiteLLM"
|
||||||
url: https://www.linkedin.com/in/krish-d/
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
|
@ -15,6 +15,7 @@ authors:
|
||||||
title: "CTO, LiteLLM"
|
title: "CTO, LiteLLM"
|
||||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "Guide to Claude Opus 4.5 and advanced features in LiteLLM: Tool Search, Programmatic Tool Calling, and Effort Parameter."
|
||||||
tags: [anthropic, claude, tool search, programmatic tool calling, effort, advanced features]
|
tags: [anthropic, claude, tool search, programmatic tool calling, effort, advanced features]
|
||||||
hide_table_of_contents: false
|
hide_table_of_contents: false
|
||||||
---
|
---
|
||||||
|
|
|
||||||
175
docs/my-website/blog/claude_code_beta_headers/index.md
Normal file
175
docs/my-website/blog/claude_code_beta_headers/index.md
Normal file
|
|
@ -0,0 +1,175 @@
|
||||||
|
---
|
||||||
|
slug: claude-code-beta-headers-incident
|
||||||
|
title: "Incident Report: Invalid beta headers with Claude Code"
|
||||||
|
date: 2026-02-16T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Sameer Kankute
|
||||||
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
tags: [incident-report, anthropic, stability]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
**Date:** February 13, 2026
|
||||||
|
**Duration:** ~3 hours
|
||||||
|
**Severity:** High
|
||||||
|
**Status:** Resolved
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Claude Code began sending unsupported Anthropic beta headers to non-Anthropic providers (Bedrock, Azure AI, Vertex AI), causing `invalid beta flag` errors. LiteLLM was forwarding all beta headers without provider-specific validation. Users experienced request failures when routing Claude Code requests through LiteLLM to these providers.
|
||||||
|
|
||||||
|
- **LLM calls to Anthropic:** No impact.
|
||||||
|
- **LLM calls to Bedrock/Azure/Vertex:** Failed with `invalid beta flag` errors when unsupported headers were present.
|
||||||
|
- **Cost tracking and routing:** No impact.
|
||||||
|
|
||||||
|
{/* truncate */}
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
Anthropic uses beta headers to enable experimental features in Claude. When Claude Code makes API requests, it includes headers like `anthropic-beta: prompt-caching-scope-2026-01-05,advanced-tool-use-2025-11-20`. However, not all providers support all Anthropic beta features.
|
||||||
|
|
||||||
|
Before this incident, LiteLLM forwarded all beta headers to all providers without validation:
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
sequenceDiagram
|
||||||
|
participant CC as Claude Code
|
||||||
|
participant LP as LiteLLM (old behavior)
|
||||||
|
participant Provider as Provider (Bedrock/Azure/Vertex)
|
||||||
|
|
||||||
|
CC->>LP: Request with beta headers
|
||||||
|
Note over CC,LP: anthropic-beta: header1,header2,header3
|
||||||
|
|
||||||
|
LP->>Provider: Forward ALL headers (no validation)
|
||||||
|
Note over LP,Provider: anthropic-beta: header1,header2,header3
|
||||||
|
|
||||||
|
Provider-->>LP: ❌ Error: invalid beta flag
|
||||||
|
LP-->>CC: Request fails
|
||||||
|
```
|
||||||
|
|
||||||
|
Requests succeeded for Anthropic (native support) but failed for other providers when Claude Code sent headers those providers didn't support.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Root cause
|
||||||
|
|
||||||
|
LiteLLM lacked provider-specific beta header validation. When Claude Code introduced new beta features or sent headers that specific providers didn't support, those headers were blindly forwarded, causing provider API errors.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Remediation
|
||||||
|
|
||||||
|
| # | Action | Status | Code |
|
||||||
|
|---|---|---|---|
|
||||||
|
| 1 | Create `anthropic_beta_headers_config.json` with provider-specific mappings | ✅ Done | [`anthropic_beta_headers_config.json`](https://github.com/BerriAI/litellm/blob/main/litellm/anthropic_beta_headers_config.json) |
|
||||||
|
| 2 | Implement strict validation: headers must be explicitly mapped to be forwarded | ✅ Done | [`litellm_logging.py`](https://github.com/BerriAI/litellm/blob/main/litellm/litellm_core_utils/litellm_logging.py) |
|
||||||
|
| 3 | Add `/reload/anthropic_beta_headers` endpoint for dynamic config updates | ✅ Done | Proxy management endpoints |
|
||||||
|
| 4 | Add `/schedule/anthropic_beta_headers_reload` for automatic periodic updates | ✅ Done | Proxy management endpoints |
|
||||||
|
| 5 | Support `LITELLM_ANTHROPIC_BETA_HEADERS_URL` for custom config sources | ✅ Done | Environment configuration |
|
||||||
|
| 6 | Support `LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS` for air-gapped deployments | ✅ Done | Environment configuration |
|
||||||
|
|
||||||
|
Now LiteLLM validates and transforms headers per-provider:
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
sequenceDiagram
|
||||||
|
participant CC as Claude Code
|
||||||
|
participant LP as LiteLLM (new behavior)
|
||||||
|
participant Config as Beta Headers Config
|
||||||
|
participant Provider as Provider (Bedrock/Azure/Vertex)
|
||||||
|
|
||||||
|
CC->>LP: Request with beta headers
|
||||||
|
Note over CC,LP: anthropic-beta: header1,header2,header3
|
||||||
|
|
||||||
|
LP->>Config: Load header mapping for provider
|
||||||
|
Config-->>LP: Returns mapping (header→value or null)
|
||||||
|
|
||||||
|
Note over LP: Validate & Transform:<br/>1. Check if header exists in mapping<br/>2. Filter out null values<br/>3. Map to provider-specific names
|
||||||
|
|
||||||
|
LP->>Provider: Request with filtered & mapped headers
|
||||||
|
Note over LP,Provider: anthropic-beta: mapped-header2<br/>(header1, header3 filtered out)
|
||||||
|
|
||||||
|
Provider-->>LP: ✅ Success response
|
||||||
|
LP-->>CC: Response
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Dynamic configuration updates
|
||||||
|
|
||||||
|
A key improvement is zero-downtime configuration updates. When Anthropic releases new beta features, users can update their configuration without restarting:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Manually trigger reload (no restart needed)
|
||||||
|
curl -X POST "https://your-proxy-url/reload/anthropic_beta_headers" \
|
||||||
|
-H "Authorization: Bearer YOUR_ADMIN_TOKEN"
|
||||||
|
|
||||||
|
# Or schedule automatic reloads every 24 hours
|
||||||
|
curl -X POST "https://your-proxy-url/schedule/anthropic_beta_headers_reload?hours=24" \
|
||||||
|
-H "Authorization: Bearer YOUR_ADMIN_TOKEN"
|
||||||
|
```
|
||||||
|
|
||||||
|
This prevents future incidents where Claude Code introduces new headers before LiteLLM configuration is updated.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Configuration format
|
||||||
|
|
||||||
|
The `anthropic_beta_headers_config.json` file maps input headers to provider-specific output headers:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"description": "Mapping of Anthropic beta headers for each provider.",
|
||||||
|
"anthropic": {
|
||||||
|
"advanced-tool-use-2025-11-20": "advanced-tool-use-2025-11-20",
|
||||||
|
"computer-use-2025-01-24": "computer-use-2025-01-24"
|
||||||
|
},
|
||||||
|
"bedrock_converse": {
|
||||||
|
"advanced-tool-use-2025-11-20": null,
|
||||||
|
"computer-use-2025-01-24": "computer-use-2025-01-24"
|
||||||
|
},
|
||||||
|
"azure_ai": {
|
||||||
|
"advanced-tool-use-2025-11-20": "advanced-tool-use-2025-11-20",
|
||||||
|
"computer-use-2025-01-24": "computer-use-2025-01-24"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Validation rules:**
|
||||||
|
1. Headers must exist in the mapping for the target provider
|
||||||
|
2. Headers with `null` values are filtered out (unsupported)
|
||||||
|
3. Header names can be transformed per-provider (e.g., Bedrock uses different names for some features)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Resolution steps for users
|
||||||
|
|
||||||
|
For users still experiencing issues, update to the latest LiteLLM version if < v1.81.11-nightly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install --upgrade litellm
|
||||||
|
```
|
||||||
|
|
||||||
|
Or manually reload the configuration without restarting:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -X POST "https://your-proxy-url/reload/anthropic_beta_headers" \
|
||||||
|
-H "Authorization: Bearer YOUR_ADMIN_TOKEN"
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Related documentation
|
||||||
|
|
||||||
|
- [Managing Anthropic Beta Headers](../proxy/sync_anthropic_beta_headers.md) - Complete configuration guide
|
||||||
|
- [`anthropic_beta_headers_config.json`](https://github.com/BerriAI/litellm/blob/main/litellm/anthropic_beta_headers_config.json) - Current configuration file
|
||||||
730
docs/my-website/blog/claude_opus_4_6/index.md
Normal file
730
docs/my-website/blog/claude_opus_4_6/index.md
Normal file
|
|
@ -0,0 +1,730 @@
|
||||||
|
---
|
||||||
|
slug: claude_opus_4_6
|
||||||
|
title: "Day 0 Support: Claude Opus 4.6"
|
||||||
|
date: 2026-02-05T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Sameer Kankute
|
||||||
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
description: "Day 0 support for Claude Opus 4.6 on LiteLLM AI Gateway - use across Anthropic, Azure, Vertex AI, and Bedrock."
|
||||||
|
tags: [anthropic, claude, opus 4.6]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
import Tabs from '@theme/Tabs';
|
||||||
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
LiteLLM now supports Claude Opus 4.6 on Day 0. Use it across Anthropic, Azure, Vertex AI, and Bedrock through the LiteLLM AI Gateway.
|
||||||
|
|
||||||
|
## Docker Image
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker pull ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.80.0-stable.opus-4-6
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage - Anthropic
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-opus-4-6
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-opus-4-6
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e ANTHROPIC_API_KEY=$ANTHROPIC_API_KEY \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.80.0-stable.opus-4-6 \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Usage - Azure
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-opus-4-6
|
||||||
|
litellm_params:
|
||||||
|
model: azure_ai/claude-opus-4-6
|
||||||
|
api_key: os.environ/AZURE_AI_API_KEY
|
||||||
|
api_base: os.environ/AZURE_AI_API_BASE # https://<resource>.services.ai.azure.com
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e AZURE_AI_API_KEY=$AZURE_AI_API_KEY \
|
||||||
|
-e AZURE_AI_API_BASE=$AZURE_AI_API_BASE \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.80.0-stable.opus-4-6 \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Usage - Vertex AI
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-opus-4-6
|
||||||
|
litellm_params:
|
||||||
|
model: vertex_ai/claude-opus-4-6
|
||||||
|
vertex_project: os.environ/VERTEX_PROJECT
|
||||||
|
vertex_location: us-east5
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e VERTEX_PROJECT=$VERTEX_PROJECT \
|
||||||
|
-e GOOGLE_APPLICATION_CREDENTIALS=/app/credentials.json \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
-v $(pwd)/credentials.json:/app/credentials.json \
|
||||||
|
ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.80.0-stable.opus-4-6 \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Usage - Bedrock
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-opus-4-6
|
||||||
|
litellm_params:
|
||||||
|
model: bedrock/anthropic.claude-opus-4-6-v1
|
||||||
|
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||||
|
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||||
|
aws_region_name: us-east-1
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \
|
||||||
|
-e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.80.0-stable.opus-4-6 \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Advanced Features
|
||||||
|
|
||||||
|
### Compaction
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
Litellm supports enabling compaction for the new claude-opus-4-6.
|
||||||
|
|
||||||
|
**Enabling Compaction**
|
||||||
|
|
||||||
|
To enable compaction, add the `context_management` parameter with the `compact_20260112` edit type:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "What is the weather in San Francisco?"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"context_management": {
|
||||||
|
"edits": [
|
||||||
|
{
|
||||||
|
"type": "compact_20260112"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"max_tokens": 100
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
All the parameters supported for context_management by anthropic are supported and can be directly added. Litellm automatically adds the `compact-2026-01-12` beta header in the request.
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
Enable compaction to reduce context size while preserving key information. LiteLLM automatically adds the `compact-2026-01-12` beta header when compaction is enabled.
|
||||||
|
|
||||||
|
:::info
|
||||||
|
**Provider Support:** Compaction is supported on Anthropic, Azure AI, and Vertex AI. It is **not supported** on Bedrock (Invoke or Converse APIs).
|
||||||
|
:::
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 4096,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Hi"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"context_management": {
|
||||||
|
"edits": [
|
||||||
|
{
|
||||||
|
"type": "compact_20260112"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
|
||||||
|
**Response with Compaction Block**
|
||||||
|
|
||||||
|
The response will include the compaction summary in `provider_specific_fields.compaction_blocks`:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"id": "chatcmpl-a6c105a3-4b25-419e-9551-c800633b6cb2",
|
||||||
|
"created": 1770357619,
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"object": "chat.completion",
|
||||||
|
"choices": [
|
||||||
|
{
|
||||||
|
"finish_reason": "length",
|
||||||
|
"index": 0,
|
||||||
|
"message": {
|
||||||
|
"content": "I don't have access to real-time data, so I can't provide the current weather in San Francisco. To get up-to-date weather information, I'd recommend checking:\n\n- **Weather websites** like weather.com, accuweather.com, or wunderground.com\n- **Search engines** – just Google \"San Francisco weather\"\n- **Weather apps** on your phone (e.g., Apple Weather, Google Weather)\n- **National",
|
||||||
|
"role": "assistant",
|
||||||
|
"provider_specific_fields": {
|
||||||
|
"compaction_blocks": [
|
||||||
|
{
|
||||||
|
"type": "compaction",
|
||||||
|
"content": "Summary of the conversation: The user requested help building a web scraper..."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"usage": {
|
||||||
|
"completion_tokens": 100,
|
||||||
|
"prompt_tokens": 86,
|
||||||
|
"total_tokens": 186
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Using Compaction Blocks in Follow-up Requests**
|
||||||
|
|
||||||
|
To continue the conversation with compaction, include the compaction block in the assistant message's `provider_specific_fields`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "How can I build a web scraper?"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"role": "assistant",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"text": "Certainly! To build a basic web scraper, you'll typically use a programming language like Python along with libraries such as `requests` (for fetching web pages) and `BeautifulSoup` (for parsing HTML). Here's a basic example:\n\n```python\nimport requests\nfrom bs4 import BeautifulSoup\n\nurl = 'https://example.com'\nresponse = requests.get(url)\nsoup = BeautifulSoup(response.text, 'html.parser')\n\n# Extract and print all text\ntext = soup.get_text()\nprint(text)\n```\n\nLet me know what you're interested in scraping or if you need help with a specific website!"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"provider_specific_fields": {
|
||||||
|
"compaction_blocks": [
|
||||||
|
{
|
||||||
|
"type": "compaction",
|
||||||
|
"content": "Summary of the conversation: The user asked how to build a web scraper, and the assistant gave an overview using Python with requests and BeautifulSoup."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "How do I use it to scrape product prices?"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"context_management": {
|
||||||
|
"edits": [
|
||||||
|
{
|
||||||
|
"type": "compact_20260112"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"max_tokens": 100
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
**Streaming Support**
|
||||||
|
|
||||||
|
Compaction blocks are also supported in streaming mode. You'll receive:
|
||||||
|
- `compaction_start` event when a compaction block begins
|
||||||
|
- `compaction_delta` events with the compaction content
|
||||||
|
- The accumulated `compaction_blocks` in `provider_specific_fields`
|
||||||
|
|
||||||
|
### Adaptive Thinking
|
||||||
|
|
||||||
|
:::note
|
||||||
|
When using `reasoning_effort` with Claude Opus 4.6, all values (`low`, `medium`, `high`) are mapped to `thinking: {type: "adaptive"}`. To use explicit thinking budgets with `type: "enabled"`, pass the native `thinking` parameter directly (see "Native thinking param" tab below).
|
||||||
|
:::
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
LiteLLM supports adaptive thinking through the `reasoning_effort` parameter:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Solve this complex problem: What is the optimal strategy for..."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"reasoning_effort": "high"
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
Use the `thinking` parameter with `type: "adaptive"` to enable adaptive thinking mode:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 16000,
|
||||||
|
"thinking": {
|
||||||
|
"type": "adaptive"
|
||||||
|
},
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Explain why the sum of two even numbers is always even."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="native" label="Native thinking param">
|
||||||
|
|
||||||
|
Use the `thinking` parameter directly for adaptive thinking via the SDK:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import litellm
|
||||||
|
|
||||||
|
response = litellm.completion(
|
||||||
|
model="anthropic/claude-opus-4-6",
|
||||||
|
messages=[{"role": "user", "content": "Solve this complex problem: What is the optimal strategy for..."}],
|
||||||
|
thinking={"type": "adaptive"},
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### Effort Levels
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
Four effort levels available: `low`, `medium`, `high` (default), and `max`. Pass directly via the `output_config` parameter:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Explain quantum computing"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_config": {
|
||||||
|
"effort": "medium"
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
You can use reasoning effort plus output_config to have more control on the model.
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
Four effort levels available: `low`, `medium`, `high` (default), and `max`. Pass directly via the `output_config` parameter:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 4096,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Explain quantum computing"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_config": {
|
||||||
|
"effort": "medium"
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### 1M Token Context (Beta)
|
||||||
|
|
||||||
|
Opus 4.6 supports 1M token context. Premium pricing applies for prompts exceeding 200k tokens ($10/$37.50 per million input/output tokens). LiteLLM supports cost calculations for 1M token contexts.
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
To use the 1M token context window, you need to forward the `anthropic-beta` header from your client to the LLM provider.
|
||||||
|
|
||||||
|
**Step 1: Enable header forwarding in your config**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
general_settings:
|
||||||
|
forward_client_headers_to_llm_api: true
|
||||||
|
```
|
||||||
|
|
||||||
|
**Step 2: Send requests with the beta header**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--header 'anthropic-beta: context-1m-2025-08-07' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Analyze this large document..."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
To use the 1M token context window, you need to forward the `anthropic-beta` header from your client to the LLM provider.
|
||||||
|
|
||||||
|
**Step 1: Enable header forwarding in your config**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
general_settings:
|
||||||
|
forward_client_headers_to_llm_api: true
|
||||||
|
```
|
||||||
|
|
||||||
|
**Step 2: Send requests with the beta header**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'anthropic-beta: context-1m-2025-08-07' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 16000,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Analyze this large document..."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
:::tip
|
||||||
|
You can combine multiple beta headers by separating them with commas:
|
||||||
|
```bash
|
||||||
|
--header 'anthropic-beta: context-1m-2025-08-07,compact-2026-01-12'
|
||||||
|
```
|
||||||
|
:::
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### US-Only Inference
|
||||||
|
|
||||||
|
Available at 1.1× token pricing. LiteLLM automatically tracks costs for US-only inference.
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
Use the `inference_geo` parameter to specify US-only inference:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "What is the capital of France?"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"inference_geo": "us"
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
LiteLLM will automatically apply the 1.1× pricing multiplier for US-only inference in cost tracking.
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
Use the `inference_geo` parameter to specify US-only inference:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 4096,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "What is the capital of France?"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"inference_geo": "us"
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
LiteLLM will automatically apply the 1.1× pricing multiplier for US-only inference in cost tracking.
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### Fast Mode
|
||||||
|
|
||||||
|
:::info
|
||||||
|
Fast mode is **only supported on the Anthropic provider** (`anthropic/claude-opus-4-6`). It is not available on Azure AI, Vertex AI, or Bedrock.
|
||||||
|
:::
|
||||||
|
|
||||||
|
**Pricing:**
|
||||||
|
- Standard: $5 input / $25 output per MTok
|
||||||
|
- Fast: $30 input / $150 output per MTok (6× premium)
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="completions" label="/chat/completions">
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Refactor this module..."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"max_tokens": 4096,
|
||||||
|
"speed": "fast"
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
**Using OpenAI SDK:**
|
||||||
|
|
||||||
|
```python
|
||||||
|
import openai
|
||||||
|
|
||||||
|
client = openai.OpenAI(
|
||||||
|
api_key="your-litellm-key",
|
||||||
|
base_url="http://0.0.0.0:4000"
|
||||||
|
)
|
||||||
|
|
||||||
|
response = client.chat.completions.create(
|
||||||
|
model="claude-opus-4-6",
|
||||||
|
messages=[{"role": "user", "content": "Refactor this module..."}],
|
||||||
|
max_tokens=4096,
|
||||||
|
extra_body={"speed": "fast"}
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Using LiteLLM SDK:**
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm import completion
|
||||||
|
|
||||||
|
response = completion(
|
||||||
|
model="anthropic/claude-opus-4-6",
|
||||||
|
messages=[{"role": "user", "content": "Refactor this module..."}],
|
||||||
|
max_tokens=4096,
|
||||||
|
speed="fast"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
LiteLLM automatically tracks the higher costs for fast mode in usage and cost calculations.
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="messages" label="/v1/messages">
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'x-api-key: sk-12345' \
|
||||||
|
--header 'content-type: application/json' \
|
||||||
|
--data '{
|
||||||
|
"model": "claude-opus-4-6",
|
||||||
|
"max_tokens": 4096,
|
||||||
|
"speed": "fast",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Refactor this module..."
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
LiteLLM automatically:
|
||||||
|
- Adds the `fast-mode-2026-02-01` beta header
|
||||||
|
- Tracks the 6× premium pricing in cost calculations
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
220
docs/my-website/blog/fastapi_middleware_performance/index.mdx
Normal file
220
docs/my-website/blog/fastapi_middleware_performance/index.mdx
Normal file
|
|
@ -0,0 +1,220 @@
|
||||||
|
---
|
||||||
|
slug: fastapi-middleware-performance
|
||||||
|
title: "Your Middleware Could Be a Bottleneck"
|
||||||
|
date: 2026-02-07T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
- name: Ryan Crabbe
|
||||||
|
title: "Performance Engineer, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/ryan-crabbe-0b9687214
|
||||||
|
image_url: https://media.licdn.com/dms/image/v2/D5603AQHt1t9Z4BJ6Gw/profile-displayphoto-shrink_400_400/profile-displayphoto-shrink_400_400/0/1724453682340?e=1772064000&v=beta&t=VXdmr13rsNB05wyA2F1TENOB5UuDHUZ0FCHTolNyR5M
|
||||||
|
description: "How we improved LiteLLM proxy latency and throughput by replacing a single middleware base class"
|
||||||
|
tags: [performance, fastapi, middleware]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
import { BaseHTTPMiddlewareAnimation, PureASGIAnimation, BenchmarkVisualization } from '@site/src/components/MiddlewareDiagrams';
|
||||||
|
|
||||||
|
> How we improved LiteLLM proxy latency and throughput by replacing a single, simple middleware base class
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Our Setup
|
||||||
|
|
||||||
|
The LiteLLM proxy server has two middleware layers. The first is Starlette's `CORSMiddleware` (re-exported by FastAPI), which is a pure ASGI middleware. Then we have a simple BaseHTTPMiddleware called PrometheusAuthMiddleware.
|
||||||
|
|
||||||
|
The job of `PrometheusAuthMiddleware` is to authenticate requests to the `/metrics` endpoint. It's not on by default, you enable it with a flag in your proxy config:
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary>Proxy config flag</summary>
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
litellm_settings:
|
||||||
|
require_auth_for_metrics_endpoint: true
|
||||||
|
```
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
The middleware checks two things: is the request hitting `/metrics`, and is auth even enabled? If both checks fail, which they do for the vast majority of requests, it just passes the request through unchanged.
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary>PrometheusAuthMiddleware source</summary>
|
||||||
|
|
||||||
|
```python
|
||||||
|
class PrometheusAuthMiddleware(BaseHTTPMiddleware):
|
||||||
|
async def dispatch(self, request: Request, call_next):
|
||||||
|
if self._is_prometheus_metrics_endpoint(request):
|
||||||
|
if self._should_run_auth_on_metrics_endpoint() is True:
|
||||||
|
try:
|
||||||
|
await user_api_key_auth(request=request, api_key=...)
|
||||||
|
except Exception as e:
|
||||||
|
return JSONResponse(status_code=401, content=...)
|
||||||
|
response = await call_next(request)
|
||||||
|
return response
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _is_prometheus_metrics_endpoint(request: Request):
|
||||||
|
if "/metrics" in request.url.path:
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
```
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
Looks harmless. Subclass `BaseHTTPMiddleware`, implement `dispatch()`, done. This is what you will see in Starlette's documentation<sup>[1](#footnote-1)</sup>.
|
||||||
|
|
||||||
|
{/* truncate */}
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What BaseHTTPMiddleware Actually Does
|
||||||
|
|
||||||
|
When you write a `dispatch()` method, you'd expect the request to flow straight through your function and out the other side. What actually happens is much more involved.
|
||||||
|
|
||||||
|
On every request, even a pure passthrough (meaning nothing happens), `BaseHTTPMiddleware` creates **7 intermediate objects and tasks**:
|
||||||
|
|
||||||
|
<BaseHTTPMiddlewareAnimation />
|
||||||
|
|
||||||
|
It wraps the request in a new object to track body state, creates a synchronization event, allocates an in-memory channel to pass messages between your middleware and the inner app, sets up a task group to manage the lifecycle, and then runs your actual route handler in a *separate background task* when you call `call_next()`. The response body then flows back through that in-memory channel, gets re-wrapped in a streaming response object, and finally reaches the caller. That's a lot.
|
||||||
|
|
||||||
|
For a middleware that for us, does nothing on 99.9% of requests, paying this cost doesn't make sense.
|
||||||
|
|
||||||
|
Compare that to a pure ASGI middleware, which we can have just check the request path and continue along.
|
||||||
|
|
||||||
|
<PureASGIAnimation />
|
||||||
|
|
||||||
|
Our middleware is doing something really simple. For the vast majority of requests it doesn't need to do anything at all but just let the request pass through. It doesn't need task groups, memory streams, or cancel scopes. It needs a function call.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Comparing Both
|
||||||
|
|
||||||
|
We replaced the `BaseHTTPMiddleware` subclass with a pure ASGI middleware. To benchmark the difference, we used Apache Bench<sup>[2](#footnote-2)</sup> to compare both configurations of LiteLLM's middleware stack: the old setup (1 pure ASGI + 1 `BaseHTTPMiddleware`) against the new setup (2 pure ASGI).
|
||||||
|
|
||||||
|
A minimal FastAPI app serves `GET /health` → `PlainTextResponse("ok")`. The endpoint does zero work to isolate the middleware overhead: any difference between configs is purely the cost of the middleware plumbing itself. Both middlewares are just calling the next layer. Same work, different base class.
|
||||||
|
|
||||||
|
Apache Bench (`ab`) fires requests at the server with 1,000 concurrent connections and a single uvicorn worker. One worker means one event loop, so the benchmark directly measures how each middleware design handles concurrent load on a single thread.
|
||||||
|
|
||||||
|
<BenchmarkVisualization />
|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary>Try it yourself</summary>
|
||||||
|
|
||||||
|
Save the script below as `benchmark_middleware.py`, then run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Terminal 1 — start the "before" server (1 ASGI + 1 BaseHTTPMiddleware)
|
||||||
|
python benchmark_middleware.py --middleware mixed
|
||||||
|
|
||||||
|
# Terminal 2 — benchmark it
|
||||||
|
ab -n 50000 -c 1000 http://localhost:8000/health
|
||||||
|
|
||||||
|
# Stop the server, then start the "after" server (2x pure ASGI)
|
||||||
|
python benchmark_middleware.py --middleware asgi
|
||||||
|
|
||||||
|
# Terminal 2 — benchmark again
|
||||||
|
ab -n 50000 -c 1000 http://localhost:8000/health
|
||||||
|
```
|
||||||
|
|
||||||
|
```python
|
||||||
|
import argparse
|
||||||
|
import uvicorn
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from fastapi.responses import PlainTextResponse
|
||||||
|
from starlette.middleware.base import BaseHTTPMiddleware
|
||||||
|
from starlette.requests import Request
|
||||||
|
from starlette.types import ASGIApp, Receive, Scope, Send
|
||||||
|
|
||||||
|
|
||||||
|
class NoOpBaseHTTPMiddleware(BaseHTTPMiddleware):
|
||||||
|
async def dispatch(self, request: Request, call_next):
|
||||||
|
return await call_next(request)
|
||||||
|
|
||||||
|
|
||||||
|
class NoOpPureASGIMiddleware:
|
||||||
|
def __init__(self, app: ASGIApp) -> None:
|
||||||
|
self.app = app
|
||||||
|
|
||||||
|
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
|
||||||
|
await self.app(scope, receive, send)
|
||||||
|
|
||||||
|
|
||||||
|
def create_app(middleware_type: str | None = None, layers: int = 2) -> FastAPI:
|
||||||
|
app = FastAPI()
|
||||||
|
|
||||||
|
@app.get("/health")
|
||||||
|
async def health():
|
||||||
|
return PlainTextResponse("ok")
|
||||||
|
|
||||||
|
if middleware_type == "mixed":
|
||||||
|
app.add_middleware(NoOpBaseHTTPMiddleware)
|
||||||
|
app.add_middleware(NoOpPureASGIMiddleware)
|
||||||
|
elif middleware_type == "asgi":
|
||||||
|
for _ in range(layers):
|
||||||
|
app.add_middleware(NoOpPureASGIMiddleware)
|
||||||
|
|
||||||
|
return app
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--middleware", choices=["asgi", "mixed"], default=None)
|
||||||
|
parser.add_argument("--layers", type=int, default=2)
|
||||||
|
parser.add_argument("--port", type=int, default=8000)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
app = create_app(middleware_type=args.middleware, layers=args.layers)
|
||||||
|
uvicorn.run(app, host="0.0.0.0", port=args.port, workers=1, log_level="warning")
|
||||||
|
```
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Our Change
|
||||||
|
|
||||||
|
Here's what we replaced it with:
|
||||||
|
|
||||||
|
```python
|
||||||
|
class PrometheusAuthMiddleware:
|
||||||
|
def __init__(self, app: ASGIApp) -> None:
|
||||||
|
self.app = app
|
||||||
|
|
||||||
|
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
|
||||||
|
if scope["type"] != "http" or "/metrics" not in scope.get("path", ""):
|
||||||
|
await self.app(scope, receive, send)
|
||||||
|
return
|
||||||
|
|
||||||
|
if litellm.require_auth_for_metrics_endpoint is True:
|
||||||
|
request = Request(scope, receive)
|
||||||
|
api_key = request.headers.get("Authorization") or ""
|
||||||
|
try:
|
||||||
|
await user_api_key_auth(request=request, api_key=api_key)
|
||||||
|
except Exception as e:
|
||||||
|
# send 401 directly via ASGI protocol
|
||||||
|
...
|
||||||
|
return
|
||||||
|
|
||||||
|
await self.app(scope, receive, send)
|
||||||
|
```
|
||||||
|
|
||||||
|
For the 99.9% of requests that aren't hitting `/metrics`, the middleware is now one dict lookup, one string check, and one function call. No objects allocated, no tasks spawned.
|
||||||
|
|
||||||
|
It's important to evaluate if the tools you're using are the right fit for the job as your software grows and handles more responsiblity. We're now putting in a static analysis check to prevent this from happening again with any newly introduced middlewares. If we find the use case is necessary then that's okay and we'll reevalute but for everything LiteLLM needs to do at the moment it's not.
|
||||||
|
|
||||||
|
This middleware change was one part of a broader optimization effort on the LiteLLM proxy. Across all optimizations combined, we've measured about a **30% reduction in proxy overhead** over the past two weeks.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
<a id="footnote-1"></a>
|
||||||
|
<sup>1</sup> [Starlette Middleware — BaseHTTPMiddleware](https://starlette.dev/middleware/#basehttpmiddleware)
|
||||||
|
|
||||||
|
<a id="footnote-2"></a>
|
||||||
|
<sup>2</sup> [Apache HTTP server benchmarking tool (`ab`)](https://httpd.apache.org/docs/2.4/programs/ab.html)
|
||||||
|
|
@ -6,7 +6,7 @@ authors:
|
||||||
- name: Sameer Kankute
|
- name: Sameer Kankute
|
||||||
title: SWE @ LiteLLM (LLM Translation)
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
url: https://www.linkedin.com/in/sameer-kankute/
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQHB_loQYd5gjg/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1719137160975?e=1765411200&v=beta&t=c8396f--_lH6Fb_pVvx_jGholPfcl0bvwmNynbNdnII
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
- name: Krrish Dholakia
|
- name: Krrish Dholakia
|
||||||
title: "CEO, LiteLLM"
|
title: "CEO, LiteLLM"
|
||||||
url: https://www.linkedin.com/in/krish-d/
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
|
@ -15,6 +15,7 @@ authors:
|
||||||
title: "CTO, LiteLLM"
|
title: "CTO, LiteLLM"
|
||||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "Common questions and best practices for using gemini-3-pro-preview with LiteLLM Proxy and SDK."
|
||||||
tags: [gemini, day 0 support, llms]
|
tags: [gemini, day 0 support, llms]
|
||||||
hide_table_of_contents: false
|
hide_table_of_contents: false
|
||||||
---
|
---
|
||||||
|
|
|
||||||
255
docs/my-website/blog/gemini_3_flash/index.md
Normal file
255
docs/my-website/blog/gemini_3_flash/index.md
Normal file
|
|
@ -0,0 +1,255 @@
|
||||||
|
---
|
||||||
|
slug: gemini_3_flash
|
||||||
|
title: "DAY 0 Support: Gemini 3 Flash on LiteLLM"
|
||||||
|
date: 2025-12-17T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Sameer Kankute
|
||||||
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "Guide to using Gemini 3 Flash on LiteLLM Proxy and SDK with day 0 support."
|
||||||
|
tags: [gemini, day 0 support, llms]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
|
||||||
|
import Tabs from '@theme/Tabs';
|
||||||
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
# Gemini 3 Flash Day 0 Support
|
||||||
|
|
||||||
|
LiteLLM now supports `gemini-3-flash-preview` and all the new API changes along with it.
|
||||||
|
|
||||||
|
:::note
|
||||||
|
If you only want cost tracking, you need no change in your current Litellm version. But if you want the support for new features introduced along with it like thinking levels, you will need to use v1.80.8-stable.1 or above.
|
||||||
|
:::
|
||||||
|
|
||||||
|
## Deploy this version
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="docker" label="Docker">
|
||||||
|
|
||||||
|
``` showLineNumbers title="docker run litellm"
|
||||||
|
docker run \
|
||||||
|
-e STORE_MODEL_IN_DB=True \
|
||||||
|
-p 4000:4000 \
|
||||||
|
ghcr.io/berriai/litellm:main-v1.80.8-stable.1
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="pip" label="Pip">
|
||||||
|
|
||||||
|
``` showLineNumbers title="pip install litellm"
|
||||||
|
pip install litellm==1.80.8.post1
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## What's New
|
||||||
|
|
||||||
|
### 1. New Thinking Levels: `thinkingLevel` with MINIMAL & MEDIUM
|
||||||
|
|
||||||
|
Gemini 3 Flash introduces granular thinking control with `thinkingLevel` instead of `thinkingBudget`.
|
||||||
|
- **MINIMAL**: Ultra-lightweight thinking for fast responses
|
||||||
|
- **MEDIUM**: Balanced thinking for complex reasoning
|
||||||
|
- **HIGH**: Maximum reasoning depth
|
||||||
|
|
||||||
|
LiteLLM automatically maps the OpenAI `reasoning_effort` parameter to Gemini's `thinkingLevel`, so you can use familiar `reasoning_effort` values (`minimal`, `low`, `medium`, `high`) without changing your code!
|
||||||
|
|
||||||
|
### 2. Thought Signatures
|
||||||
|
|
||||||
|
Like `gemini-3-pro`, this model also includes thought signatures for tool calls. LiteLLM handles signature extraction and embedding internally. [Learn more about thought signatures](../gemini_3/index.md#thought-signatures).
|
||||||
|
|
||||||
|
**Edge Case Handling**: If thought signatures are missing in the request, LiteLLM adds a dummy signature ensuring the API call doesn't break
|
||||||
|
|
||||||
|
---
|
||||||
|
## Supported Endpoints
|
||||||
|
|
||||||
|
LiteLLM provides **full end-to-end support** for Gemini 3 Flash on:
|
||||||
|
|
||||||
|
- ✅ `/v1/chat/completions` - OpenAI-compatible chat completions endpoint
|
||||||
|
- ✅ `/v1/responses` - OpenAI Responses API endpoint (streaming and non-streaming)
|
||||||
|
- ✅ [`/v1/messages`](../../docs/anthropic_unified) - Anthropic-compatible messages endpoint
|
||||||
|
- ✅ `/v1/generateContent` – [Google Gemini API](../../docs/generateContent.md) compatible endpoint
|
||||||
|
All endpoints support:
|
||||||
|
- Streaming and non-streaming responses
|
||||||
|
- Function calling with thought signatures
|
||||||
|
- Multi-turn conversations
|
||||||
|
- All Gemini 3-specific features
|
||||||
|
- Converstion of provider specific thinking related param to thinkingLevel
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="sdk" label="SDK">
|
||||||
|
|
||||||
|
**Basic Usage with MEDIUM thinking (NEW)**
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm import completion
|
||||||
|
|
||||||
|
# No need to make any changes to your code as we map openai reasoning param to thinkingLevel
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "Solve this complex math problem: 25 * 4 + 10"}],
|
||||||
|
reasoning_effort="medium", # NEW: MEDIUM thinking level
|
||||||
|
)
|
||||||
|
|
||||||
|
print(response.choices[0].message.content)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="proxy" label="PROXY">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: gemini-3-flash
|
||||||
|
litellm_params:
|
||||||
|
model: gemini/gemini-3-flash-preview
|
||||||
|
api_key: os.environ/GEMINI_API_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Call with MEDIUM thinking**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -X POST http://localhost:4000/v1/chat/completions \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||||
|
-d '{
|
||||||
|
"model": "gemini-3-flash",
|
||||||
|
"messages": [{"role": "user", "content": "Complex reasoning task"}],
|
||||||
|
"reasoning_effort": "medium"
|
||||||
|
}'
|
||||||
|
``'
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## All `reasoning_effort` Levels
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="minimal" label="MINIMAL">
|
||||||
|
|
||||||
|
**Ultra-fast, minimal reasoning**
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm import completion
|
||||||
|
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "What's 2+2?"}],
|
||||||
|
reasoning_effort="minimal",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="low" label="LOW">
|
||||||
|
|
||||||
|
**Simple instruction following**
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "Write a haiku about coding"}],
|
||||||
|
reasoning_effort="low",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="medium" label="MEDIUM (NEW)">
|
||||||
|
|
||||||
|
**Balanced reasoning for complex tasks** ✨
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "Analyze this dataset and find patterns"}],
|
||||||
|
reasoning_effort="medium", # NEW!
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="high" label="HIGH">
|
||||||
|
|
||||||
|
**Maximum reasoning depth**
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "Prove this mathematical theorem"}],
|
||||||
|
reasoning_effort="high",
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key Features
|
||||||
|
|
||||||
|
✅ **Thinking Levels**: MINIMAL, LOW, MEDIUM, HIGH
|
||||||
|
✅ **Thought Signatures**: Track reasoning with unique identifiers
|
||||||
|
✅ **Seamless Integration**: Works with existing OpenAI-compatible client
|
||||||
|
✅ **Backward Compatible**: Gemini 2.5 models continue using `thinkingBudget`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install litellm --upgrade
|
||||||
|
```
|
||||||
|
|
||||||
|
```python
|
||||||
|
import litellm
|
||||||
|
from litellm import completion
|
||||||
|
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-3-flash-preview",
|
||||||
|
messages=[{"role": "user", "content": "Your question here"}],
|
||||||
|
reasoning_effort="medium", # Use MEDIUM thinking
|
||||||
|
)
|
||||||
|
print(response)
|
||||||
|
```
|
||||||
|
|
||||||
|
:::note
|
||||||
|
If using this model via vertex_ai, keep the location as global as this is the only supported location as of now.
|
||||||
|
:::
|
||||||
|
|
||||||
|
|
||||||
|
## `reasoning_effort` Mapping for Gemini 3+
|
||||||
|
|
||||||
|
| reasoning_effort | thinking_level |
|
||||||
|
|------------------|----------------|
|
||||||
|
| `minimal` | `minimal` |
|
||||||
|
| `low` | `low` |
|
||||||
|
| `medium` | `medium` |
|
||||||
|
| `high` | `high` |
|
||||||
|
| `disable` | `minimal` |
|
||||||
|
| `none` | `minimal` |
|
||||||
|
|
||||||
136
docs/my-website/blog/litellm_observatory/index.md
Normal file
136
docs/my-website/blog/litellm_observatory/index.md
Normal file
|
|
@ -0,0 +1,136 @@
|
||||||
|
---
|
||||||
|
slug: litellm-observatory
|
||||||
|
title: "Improve release stability with 24 hour load tests"
|
||||||
|
date: 2026-02-06T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Alexsander Hamir
|
||||||
|
title: "Performance Engineer, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/alexsander-baptista/
|
||||||
|
image_url: https://github.com/AlexsanderHamir.png
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "How we built a long-running, release-validation system to catch regressions before they reach users."
|
||||||
|
tags: [testing, observability, reliability, releases]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
# Improve release stability with 24 hour load tests
|
||||||
|
|
||||||
|
As LiteLLM adoption has grown, so have expectations around reliability, performance, and operational safety. Meeting those expectations requires more than correctness-focused tests, it requires validating how the system behaves over time, under real-world conditions.
|
||||||
|
|
||||||
|
This post introduces **LiteLLM Observatory**, a long-running release-validation system we built to catch regressions before they reach users.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Why We Built the Observatory
|
||||||
|
|
||||||
|
LiteLLM operates at the intersection of external providers, long-lived network connections, and high-throughput workloads. While our unit and integration tests do an excellent job validating correctness, they are not designed to surface issues that only appear after extended operation.
|
||||||
|
|
||||||
|
A subtle lifecycle edge case discovered in v1.81.3 reinforced the need for stronger release validation in this area.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## A Real-World Lifecycle Edge Case
|
||||||
|
|
||||||
|
In v1.81.3, we shipped a fix for an HTTP client memory leak. The change passed unit and integration tests and behaved correctly in short-lived runs.
|
||||||
|
|
||||||
|
The issue that surfaced was not caused by a single incorrect line of logic, but by how multiple components interacted over time:
|
||||||
|
|
||||||
|
- A cached `httpx` client was configured with a 1-hour TTL
|
||||||
|
- When the cache expired, the underlying HTTP connection was closed as expected
|
||||||
|
- A higher-level client continued to hold a reference to that connection
|
||||||
|
- Subsequent requests failed with:
|
||||||
|
|
||||||
|
```
|
||||||
|
Cannot send a request, as the client has been closed
|
||||||
|
```
|
||||||
|
|
||||||
|
**Before (with bug):**
|
||||||
|
|
||||||
|
| Provider | Requests | Success | Failures | Fail % |
|
||||||
|
|----------|----------|---------|----------|--------|
|
||||||
|
| OpenAI | 720,000 | 432,000 | 288,000 | 40% |
|
||||||
|
| Azure | 692,000 | 415,200 | 276,800 | 40% |
|
||||||
|
|
||||||
|
**After (fixed):**
|
||||||
|
|
||||||
|
| Provider | Requests | Success | Failures | Fail % |
|
||||||
|
|----------|------------|-----------|----------|---------|
|
||||||
|
| OpenAI | 1,200,000 | 1,199,988 | 12 | 0.001% |
|
||||||
|
| Azure | 1,150,000 | 1,149,982 | 18 | 0.002% |
|
||||||
|
|
||||||
|
Our focus moving forward is on being the first to detect issues, even when they aren’t covered by unit tests. LiteLLM Observatory is designed to surface latency regressions, OOMs, and failure modes that only appear under real traffic patterns in **our own production deployments** during release validation.
|
||||||
|
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### How the Observatory Works
|
||||||
|
|
||||||
|
[LiteLLM Observatory](https://github.com/BerriAI/litellm-observatory) is a testing service that runs long-running tests against our LiteLLM deployments. We trigger tests by sending API requests, and results are automatically sent to Slack when tests complete.
|
||||||
|
|
||||||
|
#### How Tests Run
|
||||||
|
|
||||||
|
1. **Start a Test**: We send a request to the Observatory API with:
|
||||||
|
- Which LiteLLM deployment to test (URL and API key)
|
||||||
|
- Which test to run (e.g., `TestOAIAzureRelease`)
|
||||||
|
- Test settings (which models to test, how long to run, failure thresholds)
|
||||||
|
|
||||||
|
2. **Smart Queueing**:
|
||||||
|
- The system checks whether we are attempting to run the exact same test more than once
|
||||||
|
- If a duplicate test is already running or queued, we receive an error to avoid wasting resources
|
||||||
|
- Otherwise, the test is added to a queue and runs when capacity is available (up to 5 tests can run concurrently by default)
|
||||||
|
|
||||||
|
3. **Instant Response**: The API responds immediately—we do not wait for the test to finish. Tests may run for hours, but the request itself completes in milliseconds.
|
||||||
|
|
||||||
|
4. **Background Execution**:
|
||||||
|
- The test runs in the background, issuing requests against our LiteLLM deployment
|
||||||
|
- It tracks request success and failure rates over time
|
||||||
|
- When the test completes, results are automatically posted to our Slack channel
|
||||||
|
|
||||||
|
#### Example: The OpenAI / Azure Reliability Test
|
||||||
|
|
||||||
|
The `TestOAIAzureRelease` test is designed to catch a class of bugs that only surface after sustained runtime:
|
||||||
|
|
||||||
|
- **Duration**: Runs continuously for 3 hours
|
||||||
|
- **Behavior**: Cycles through specified models (such as `gpt-4` and `gpt-3.5-turbo`), issuing requests continuously
|
||||||
|
- **Why 3 Hours**: This helps catch issues where HTTP clients degrade or fail after extended use (for example, a bug observed in LiteLLM v1.81.3)
|
||||||
|
- **Pass / Fail Criteria**: The test passes if fewer than 1% of requests fail. If the failure rate exceeds 1%, the test fails and we are notified in Slack
|
||||||
|
- **Key Detail**: The same HTTP client is reused for the entire run, allowing us to detect lifecycle-related bugs that only appear under prolonged reuse
|
||||||
|
|
||||||
|
#### When We Use It
|
||||||
|
|
||||||
|
- **Before Deployments**: Run tests before promoting a new LiteLLM version to production
|
||||||
|
- **Routine Validation**: Schedule regular runs (daily or weekly) to catch regressions early
|
||||||
|
- **Issue Investigation**: Run tests on demand when we suspect a deployment issue
|
||||||
|
- **Long-Running Failure Detection**: Identify bugs that only appear under sustained load, beyond what short smoke tests can reveal
|
||||||
|
|
||||||
|
|
||||||
|
### Complementing Unit Tests
|
||||||
|
|
||||||
|
Unit tests remain a foundational part of our development process. They are fast and precise, but they don’t cover:
|
||||||
|
|
||||||
|
- Real provider behavior
|
||||||
|
- Long-lived network interactions
|
||||||
|
- Resource lifecycle edge cases
|
||||||
|
- Time-dependent regressions
|
||||||
|
|
||||||
|
LiteLLM Observatory complements unit tests by validating the system as it actually runs in production-like environments.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Looking Ahead
|
||||||
|
|
||||||
|
Reliability is an ongoing investment.
|
||||||
|
|
||||||
|
LiteLLM Observatory is one of several systems we’re building to continuously raise the bar on release quality and operational safety. As LiteLLM evolves, so will our validation tooling, informed by real-world usage and lessons learned.
|
||||||
|
|
||||||
|
We’ll continue to share those improvements openly as we go.
|
||||||
|
|
||||||
394
docs/my-website/blog/minimax_m2_5/index.md
Normal file
394
docs/my-website/blog/minimax_m2_5/index.md
Normal file
|
|
@ -0,0 +1,394 @@
|
||||||
|
---
|
||||||
|
slug: minimax_m2_5
|
||||||
|
title: "Day 0 Support: MiniMax-M2.5"
|
||||||
|
date: 2026-02-12T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Sameer Kankute
|
||||||
|
title: SWE @ LiteLLM (LLM Translation)
|
||||||
|
url: https://www.linkedin.com/in/sameer-kankute/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/2001352686994907136/ONgNuSk5_400x400.jpg
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "Day 0 support for MiniMax-M2.5 on LiteLLM"
|
||||||
|
tags: [minimax, M2.5, llm]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
import Tabs from '@theme/Tabs';
|
||||||
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
LiteLLM now supports MiniMax-M2.5 on Day 0. Use it across OpenAI-compatible and Anthropic-compatible APIs through the LiteLLM AI Gateway.
|
||||||
|
|
||||||
|
## Supported Models
|
||||||
|
|
||||||
|
LiteLLM supports the following MiniMax models:
|
||||||
|
|
||||||
|
| Model | Description | Input Cost | Output Cost | Context Window |
|
||||||
|
|-------|-------------|------------|-------------|----------------|
|
||||||
|
| **MiniMax-M2.5** | Advanced reasoning, Agentic capabilities | $0.3/M tokens | $1.2/M tokens | 1M tokens |
|
||||||
|
| **MiniMax-M2.5-lightning** | Faster and More Agile (~100 tps) | $0.3/M tokens | $2.4/M tokens | 1M tokens |
|
||||||
|
|
||||||
|
## Features Supported
|
||||||
|
|
||||||
|
- **Prompt Caching**: Reduce costs with cached prompts ($0.03/M tokens for cache read, $0.375/M tokens for cache write)
|
||||||
|
- **Function Calling**: Built-in tool calling support
|
||||||
|
- **Reasoning**: Advanced reasoning capabilities with thinking support
|
||||||
|
- **System Messages**: Full system message support
|
||||||
|
- **Cost Tracking**: Automatic cost calculation for all requests
|
||||||
|
|
||||||
|
## Docker Image
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker pull litellm/litellm:v1.81.3-stable
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage - OpenAI Compatible API (/v1/chat/completions)
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: minimax-m2-5
|
||||||
|
litellm_params:
|
||||||
|
model: minimax/MiniMax-M2.5
|
||||||
|
api_key: os.environ/MINIMAX_API_KEY
|
||||||
|
api_base: https://api.minimax.io/v1
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e MINIMAX_API_KEY=$MINIMAX_API_KEY \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
ghcr.io/berriai/litellm:v1.81.3-stable \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "minimax-m2-5",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### With Reasoning Split
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "minimax-m2-5",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Solve: 2+2=?"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"extra_body": {
|
||||||
|
"reasoning_split": true
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage - Anthropic Compatible API (/v1/messages)
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||||
|
|
||||||
|
**1. Setup config.yaml**
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: minimax-m2-5
|
||||||
|
litellm_params:
|
||||||
|
model: minimax/MiniMax-M2.5
|
||||||
|
api_key: os.environ/MINIMAX_API_KEY
|
||||||
|
api_base: https://api.minimax.io/anthropic/v1/messages
|
||||||
|
```
|
||||||
|
|
||||||
|
**2. Start the proxy**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run -d \
|
||||||
|
-p 4000:4000 \
|
||||||
|
-e MINIMAX_API_KEY=$MINIMAX_API_KEY \
|
||||||
|
-v $(pwd)/config.yaml:/app/config.yaml \
|
||||||
|
ghcr.io/berriai/litellm:v1.81.3-stable \
|
||||||
|
--config /app/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
**3. Test it!**
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "minimax-m2-5",
|
||||||
|
"max_tokens": 1000,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "what llm are you"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### With Thinking
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl --location 'http://0.0.0.0:4000/v1/messages' \
|
||||||
|
--header 'Content-Type: application/json' \
|
||||||
|
--header 'Authorization: Bearer $LITELLM_KEY' \
|
||||||
|
--data '{
|
||||||
|
"model": "minimax-m2-5",
|
||||||
|
"max_tokens": 1000,
|
||||||
|
"thinking": {
|
||||||
|
"type": "enabled",
|
||||||
|
"budget_tokens": 1000
|
||||||
|
},
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Solve: 2+2=?"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage - LiteLLM SDK
|
||||||
|
|
||||||
|
### OpenAI-compatible API
|
||||||
|
|
||||||
|
```python
|
||||||
|
import litellm
|
||||||
|
|
||||||
|
response = litellm.completion(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[
|
||||||
|
{"role": "user", "content": "Hello, how are you?"}
|
||||||
|
],
|
||||||
|
api_key="your-minimax-api-key",
|
||||||
|
api_base="https://api.minimax.io/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
print(response.choices[0].message.content)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Anthropic-compatible API
|
||||||
|
|
||||||
|
```python
|
||||||
|
import litellm
|
||||||
|
|
||||||
|
response = litellm.anthropic.messages.acreate(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||||
|
api_key="your-minimax-api-key",
|
||||||
|
api_base="https://api.minimax.io/anthropic/v1/messages",
|
||||||
|
max_tokens=1000
|
||||||
|
)
|
||||||
|
|
||||||
|
print(response.choices[0].message.content)
|
||||||
|
```
|
||||||
|
|
||||||
|
### With Thinking
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = litellm.anthropic.messages.acreate(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[{"role": "user", "content": "Solve: 2+2=?"}],
|
||||||
|
thinking={"type": "enabled", "budget_tokens": 1000},
|
||||||
|
api_key="your-minimax-api-key"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Access thinking content
|
||||||
|
for block in response.choices[0].message.content:
|
||||||
|
if hasattr(block, 'type') and block.type == 'thinking':
|
||||||
|
print(f"Thinking: {block.thinking}")
|
||||||
|
```
|
||||||
|
|
||||||
|
### With Reasoning Split (OpenAI API)
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = litellm.completion(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[
|
||||||
|
{"role": "user", "content": "Solve: 2+2=?"}
|
||||||
|
],
|
||||||
|
extra_body={"reasoning_split": True},
|
||||||
|
api_key="your-minimax-api-key",
|
||||||
|
api_base="https://api.minimax.io/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Access thinking and response
|
||||||
|
if hasattr(response.choices[0].message, 'reasoning_details'):
|
||||||
|
print(f"Thinking: {response.choices[0].message.reasoning_details}")
|
||||||
|
print(f"Response: {response.choices[0].message.content}")
|
||||||
|
```
|
||||||
|
|
||||||
|
## Cost Tracking
|
||||||
|
|
||||||
|
LiteLLM automatically tracks costs for MiniMax-M2.5 requests. The pricing is:
|
||||||
|
|
||||||
|
- **Input**: $0.3 per 1M tokens
|
||||||
|
- **Output**: $1.2 per 1M tokens
|
||||||
|
- **Cache Read**: $0.03 per 1M tokens
|
||||||
|
- **Cache Write**: $0.375 per 1M tokens
|
||||||
|
|
||||||
|
### Accessing Cost Information
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = litellm.completion(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[{"role": "user", "content": "Hello!"}],
|
||||||
|
api_key="your-minimax-api-key"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Access cost information
|
||||||
|
print(f"Cost: ${response._hidden_params.get('response_cost', 0)}")
|
||||||
|
```
|
||||||
|
|
||||||
|
## Streaming Support
|
||||||
|
|
||||||
|
### OpenAI API
|
||||||
|
|
||||||
|
```python
|
||||||
|
response = litellm.completion(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[{"role": "user", "content": "Tell me a story"}],
|
||||||
|
stream=True,
|
||||||
|
api_key="your-minimax-api-key",
|
||||||
|
api_base="https://api.minimax.io/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
for chunk in response:
|
||||||
|
if chunk.choices[0].delta.content:
|
||||||
|
print(chunk.choices[0].delta.content, end="")
|
||||||
|
```
|
||||||
|
|
||||||
|
### Streaming with Reasoning Split
|
||||||
|
|
||||||
|
```python
|
||||||
|
stream = litellm.completion(
|
||||||
|
model="minimax/MiniMax-M2.5",
|
||||||
|
messages=[
|
||||||
|
{"role": "user", "content": "Tell me a story"},
|
||||||
|
],
|
||||||
|
extra_body={"reasoning_split": True},
|
||||||
|
stream=True,
|
||||||
|
api_key="your-minimax-api-key",
|
||||||
|
api_base="https://api.minimax.io/v1"
|
||||||
|
)
|
||||||
|
|
||||||
|
reasoning_buffer = ""
|
||||||
|
text_buffer = ""
|
||||||
|
|
||||||
|
for chunk in stream:
|
||||||
|
if hasattr(chunk.choices[0].delta, "reasoning_details") and chunk.choices[0].delta.reasoning_details:
|
||||||
|
for detail in chunk.choices[0].delta.reasoning_details:
|
||||||
|
if "text" in detail:
|
||||||
|
reasoning_text = detail["text"]
|
||||||
|
new_reasoning = reasoning_text[len(reasoning_buffer):]
|
||||||
|
if new_reasoning:
|
||||||
|
print(new_reasoning, end="", flush=True)
|
||||||
|
reasoning_buffer = reasoning_text
|
||||||
|
|
||||||
|
if chunk.choices[0].delta.content:
|
||||||
|
content_text = chunk.choices[0].delta.content
|
||||||
|
new_text = content_text[len(text_buffer):] if text_buffer else content_text
|
||||||
|
if new_text:
|
||||||
|
print(new_text, end="", flush=True)
|
||||||
|
text_buffer = content_text
|
||||||
|
```
|
||||||
|
|
||||||
|
## Using with Native SDKs
|
||||||
|
|
||||||
|
### Anthropic SDK via LiteLLM Proxy
|
||||||
|
|
||||||
|
```python
|
||||||
|
import os
|
||||||
|
os.environ["ANTHROPIC_BASE_URL"] = "http://localhost:4000"
|
||||||
|
os.environ["ANTHROPIC_API_KEY"] = "sk-1234" # Your LiteLLM proxy key
|
||||||
|
|
||||||
|
import anthropic
|
||||||
|
|
||||||
|
client = anthropic.Anthropic()
|
||||||
|
|
||||||
|
message = client.messages.create(
|
||||||
|
model="minimax-m2-5",
|
||||||
|
max_tokens=1000,
|
||||||
|
system="You are a helpful assistant.",
|
||||||
|
messages=[
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"text": "Hi, how are you?"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
for block in message.content:
|
||||||
|
if block.type == "thinking":
|
||||||
|
print(f"Thinking:\n{block.thinking}\n")
|
||||||
|
elif block.type == "text":
|
||||||
|
print(f"Text:\n{block.text}\n")
|
||||||
|
```
|
||||||
|
|
||||||
|
### OpenAI SDK via LiteLLM Proxy
|
||||||
|
|
||||||
|
```python
|
||||||
|
import os
|
||||||
|
os.environ["OPENAI_BASE_URL"] = "http://localhost:4000"
|
||||||
|
os.environ["OPENAI_API_KEY"] = "sk-1234" # Your LiteLLM proxy key
|
||||||
|
|
||||||
|
from openai import OpenAI
|
||||||
|
|
||||||
|
client = OpenAI()
|
||||||
|
|
||||||
|
response = client.chat.completions.create(
|
||||||
|
model="minimax-m2-5",
|
||||||
|
messages=[
|
||||||
|
{"role": "system", "content": "You are a helpful assistant."},
|
||||||
|
{"role": "user", "content": "Hi, how are you?"},
|
||||||
|
],
|
||||||
|
extra_body={"reasoning_split": True},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Access thinking and response
|
||||||
|
if hasattr(response.choices[0].message, 'reasoning_details'):
|
||||||
|
print(f"Thinking:\n{response.choices[0].message.reasoning_details[0]['text']}\n")
|
||||||
|
print(f"Text:\n{response.choices[0].message.content}\n")
|
||||||
|
```
|
||||||
95
docs/my-website/blog/model_cost_map_incident/index.md
Normal file
95
docs/my-website/blog/model_cost_map_incident/index.md
Normal file
|
|
@ -0,0 +1,95 @@
|
||||||
|
---
|
||||||
|
slug: model-cost-map-incident
|
||||||
|
title: "Incident Report: Invalid model cost map on main"
|
||||||
|
date: 2026-02-10T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Ishaan Jaffer
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/ishaanjaffer/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
tags: [incident-report, stability]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|
**Date:** January 27, 2026
|
||||||
|
**Duration:** ~20 minutes
|
||||||
|
**Severity:** Low
|
||||||
|
**Status:** Resolved
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
A malformed JSON entry in `model_prices_and_context_window.json` was merged to `main` ([`562f0a0`](https://github.com/BerriAI/litellm/commit/562f0a028251750e3d75386bee0e630d9796d0df)). This caused LiteLLM to silently fall back to a stale local copy of the model cost map. Users on older package versions lost cost tracking for newer models only (e.g. `azure/gpt-5.2`). No LLM calls were blocked.
|
||||||
|
|
||||||
|
- **LLM calls and proxy routing:** No impact.
|
||||||
|
- **Cost tracking:** Impacted for newer models not present in the local backup. Older models were unaffected. The incident lasted ~20 minutes until the commit was reverted.
|
||||||
|
|
||||||
|
{/* truncate */}
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The model cost map is not in the request path. It is used after the LLM response comes back, inside a try/catch, to calculate spend. A missing entry never blocks a call.
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
flowchart TD
|
||||||
|
A["1. litellm.completion() receives request
|
||||||
|
litellm/main.py"] --> B["2. Route to provider
|
||||||
|
litellm/litellm_core_utils/get_llm_provider_logic.py"]
|
||||||
|
B --> C["3. LLM returns response
|
||||||
|
litellm/main.py"]
|
||||||
|
C --> D["4. Post-call: look up model in cost map
|
||||||
|
litellm/cost_calculator.py"]
|
||||||
|
D -->|"found"| E["5a. Attach cost to response"]
|
||||||
|
D -->|"not found (try/catch)"| F["5b. Log warning, set cost=0"]
|
||||||
|
E --> G["6. Return response to caller"]
|
||||||
|
F --> G
|
||||||
|
|
||||||
|
style D fill:#fff3cd,stroke:#ffc107
|
||||||
|
style F fill:#fff3cd,stroke:#ffc107
|
||||||
|
style E fill:#d4edda,stroke:#28a745
|
||||||
|
style G fill:#d4edda,stroke:#28a745
|
||||||
|
```
|
||||||
|
|
||||||
|
Both paths return a response to the caller. When the cost map lookup fails, the only difference is `cost=0` on that request.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Root cause
|
||||||
|
|
||||||
|
LiteLLM fetches the model cost map from GitHub `main` at import time. If the fetch fails, it falls back to a local backup bundled with the package. Before this incident, the fallback was completely silent -- no warning was logged.
|
||||||
|
|
||||||
|
A contributor PR introduced an extra `{` bracket, producing invalid JSON. The remote fetch failed with `JSONDecodeError`, triggering the silent fallback. Users on older package versions had backup files missing newer models.
|
||||||
|
|
||||||
|
**Timeline:**
|
||||||
|
|
||||||
|
1. Malformed JSON merged to `main`
|
||||||
|
2. LiteLLM installations fall back to local backup on next import
|
||||||
|
3. Users report `"This model isn't mapped yet"` for newer models
|
||||||
|
4. Bad commit identified and reverted (~20 minutes)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Remediation
|
||||||
|
|
||||||
|
| # | Action | Status | Code |
|
||||||
|
|---|---|---|---|
|
||||||
|
| 1 | CI validation on `model_prices_and_context_window.json` | ✅ Done | [`test-model-map.yaml`](https://github.com/BerriAI/litellm/blob/main/.github/workflows/test-model-map.yaml) |
|
||||||
|
| 2 | Warning log on fallback to local backup | ✅ Done | [`get_model_cost_map.py#L57-L68`](https://github.com/BerriAI/litellm/blob/main/litellm/litellm_core_utils/get_model_cost_map.py#L57-L68) |
|
||||||
|
| 3 | `GetModelCostMap` class with integrity validation helpers | ✅ Done | [`get_model_cost_map.py#L24-L149`](https://github.com/BerriAI/litellm/blob/main/litellm/litellm_core_utils/get_model_cost_map.py#L24-L149) |
|
||||||
|
| 4 | Resilience test suite (bad hosted map, fallback, completion) | ✅ Done | [`test_model_cost_map_resilience.py#L150-L291`](https://github.com/BerriAI/litellm/blob/main/tests/llm_translation/test_model_cost_map_resilience.py#L150-L291) |
|
||||||
|
| 5 | Test that backup model cost map always exists and contains common models | ✅ Done | [`test_model_cost_map_resilience.py#L213-L228`](https://github.com/BerriAI/litellm/blob/main/tests/llm_translation/test_model_cost_map_resilience.py#L213-L228) |
|
||||||
|
|
||||||
|
Enterprises that require zero external dependencies at import time can set `LITELLM_LOCAL_MODEL_COST_MAP=True` to skip the GitHub fetch entirely.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Other dependencies on external resources
|
||||||
|
|
||||||
|
| Dependency | Impact if unavailable | Fallback |
|
||||||
|
|---|---|---|
|
||||||
|
| Model cost map (GitHub) | Cost tracking for newer models | Local backup (now with warning) |
|
||||||
|
| JWT public keys (IDP/SSO) | Auth fails | None |
|
||||||
|
| OIDC UserInfo (IDP/SSO) | Auth fails | None |
|
||||||
|
| HuggingFace model API | HF provider calls fail | None |
|
||||||
|
| Ollama tags (localhost) | Ollama model list stale | Static list |
|
||||||
92
docs/my-website/blog/sub_millisecond_proxy_overhead/index.md
Normal file
92
docs/my-website/blog/sub_millisecond_proxy_overhead/index.md
Normal file
|
|
@ -0,0 +1,92 @@
|
||||||
|
---
|
||||||
|
slug: sub-millisecond-proxy-overhead
|
||||||
|
title: "Achieving Sub-Millisecond Proxy Overhead"
|
||||||
|
date: 2026-02-02T10:00:00
|
||||||
|
authors:
|
||||||
|
- name: Alexsander Hamir
|
||||||
|
title: "Performance Engineer, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/alexsander-baptista/
|
||||||
|
image_url: https://github.com/AlexsanderHamir.png
|
||||||
|
- name: Krrish Dholakia
|
||||||
|
title: "CEO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/krish-d/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
|
||||||
|
- name: Ishaan Jaff
|
||||||
|
title: "CTO, LiteLLM"
|
||||||
|
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||||
|
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||||
|
description: "Our Q1 performance target and architectural direction for achieving sub-millisecond proxy overhead on modest hardware."
|
||||||
|
tags: [performance, architecture]
|
||||||
|
hide_table_of_contents: false
|
||||||
|
---
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
# Achieving Sub-Millisecond Proxy Overhead
|
||||||
|
|
||||||
|
## Introduction
|
||||||
|
|
||||||
|
Our Q1 performance target is to aggressively move toward sub-millisecond proxy overhead on a single instance with 4 CPUs and 8 GB of RAM, and to continue pushing that boundary over time. Our broader goal is to make LiteLLM inexpensive to deploy, lightweight, and fast. This post outlines the architectural direction behind that effort.
|
||||||
|
|
||||||
|
Proxy overhead refers to the latency introduced by LiteLLM itself, independent of the upstream provider.
|
||||||
|
|
||||||
|
To measure it, we run the same workload directly against the provider and through LiteLLM at identical QPS (for example, 1,000 QPS) and compare the latency delta. To reduce noise, the load generator, LiteLLM, and a mock LLM endpoint all run on the same machine, ensuring the difference reflects proxy overhead rather than network latency.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Where We're Coming From
|
||||||
|
|
||||||
|
Under the same benchmark originally conducted by [TensorZero](https://www.tensorzero.com/docs/gateway/benchmarks), LiteLLM previously failed at around 1,000 QPS.
|
||||||
|
|
||||||
|
That is no longer the case. Today, LiteLLM can be stress-tested at 1,000 QPS with no failures and can scale up to 5,000 QPS without failures on a 4-CPU, 8-GB RAM single instance setup.
|
||||||
|
|
||||||
|
This establishes a more up to date baseline and provides useful context as we continue working on proxy overhead and overall performance.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Design Choice
|
||||||
|
|
||||||
|
Achieving sub-millisecond proxy overhead with a Python-based system requires being deliberate about where work happens.
|
||||||
|
|
||||||
|
Python is a strong fit for flexibility and extensibility: provider abstraction, configuration-driven routing, and a rich callback ecosystem. These are areas where development velocity and correctness matter more than raw throughput.
|
||||||
|
|
||||||
|
At higher request rates, however, certain classes of work become expensive when executed inside the Python process on every request. Rather than rewriting LiteLLM or introducing complex deployment requirements, we adopt an optional **sidecar architecture**.
|
||||||
|
|
||||||
|
This architectural change is how we intend to make LiteLLM **permanently fast**. While it supports our near-term performance targets, it is a long-term investment.
|
||||||
|
|
||||||
|
Python continues to own:
|
||||||
|
|
||||||
|
- Request validation and normalization
|
||||||
|
- Model and provider selection
|
||||||
|
- Callbacks and integrations
|
||||||
|
|
||||||
|
The sidecar owns **performance-critical execution**, such as:
|
||||||
|
|
||||||
|
- Efficient request forwarding
|
||||||
|
- Connection reuse and pooling
|
||||||
|
- Enforcing timeouts and limits
|
||||||
|
- Aggregating high-frequency metrics
|
||||||
|
|
||||||
|
This separation allows each component to focus on what it does best: Python acts as the control plane, while the sidecar handles the hot path.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Why the Sidecar Is Optional
|
||||||
|
|
||||||
|
The sidecar is intentionally **optional**.
|
||||||
|
|
||||||
|
This allows us to ship it incrementally, validate it under real-world workloads, and avoid making it a hard dependency before it is fully battle-tested across all LiteLLM features.
|
||||||
|
|
||||||
|
Just as importantly, this ensures that self-hosting LiteLLM remains simple. The sidecar is bundled and started automatically, requires no additional infrastructure, and can be disabled entirely. From a user's perspective, LiteLLM continues to behave like a single service.
|
||||||
|
|
||||||
|
As of today, the sidecar is an optimization, not a requirement.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Conclusion
|
||||||
|
|
||||||
|
Sub-millisecond proxy overhead is not achieved through a single optimization, but through architectural changes.
|
||||||
|
|
||||||
|
By keeping Python focused on orchestration and extensibility, and offloading performance-critical execution to a sidecar, we establish a foundation for making LiteLLM **permanently fast over time**—even on modest hardware such as a 1-CPU, 2-GB RAM instance, while keeping deployment and self-hosting simple.
|
||||||
|
|
||||||
|
This work extends beyond Q1, and we will continue sharing benchmarks and updates as the architecture evolves.
|
||||||
|
|
@ -16,10 +16,12 @@ Add A2A Agents on LiteLLM AI Gateway, Invoke agents in A2A Protocol, track reque
|
||||||
|
|
||||||
| Feature | Supported |
|
| Feature | Supported |
|
||||||
|---------|-----------|
|
|---------|-----------|
|
||||||
|
| Supported Agent Providers | A2A, Vertex AI Agent Engine, LangGraph, Azure AI Foundry, Bedrock AgentCore, Pydantic AI |
|
||||||
| Logging | ✅ |
|
| Logging | ✅ |
|
||||||
| Load Balancing | ✅ |
|
| Load Balancing | ✅ |
|
||||||
| Streaming | ✅ |
|
| Streaming | ✅ |
|
||||||
|
|
||||||
|
|
||||||
:::tip
|
:::tip
|
||||||
|
|
||||||
LiteLLM follows the [A2A (Agent-to-Agent) Protocol](https://github.com/google/A2A) for invoking agents.
|
LiteLLM follows the [A2A (Agent-to-Agent) Protocol](https://github.com/google/A2A) for invoking agents.
|
||||||
|
|
@ -28,6 +30,8 @@ LiteLLM follows the [A2A (Agent-to-Agent) Protocol](https://github.com/google/A2
|
||||||
|
|
||||||
## Adding your Agent
|
## Adding your Agent
|
||||||
|
|
||||||
|
### Add A2A Agents
|
||||||
|
|
||||||
You can add A2A-compatible agents through the LiteLLM Admin UI.
|
You can add A2A-compatible agents through the LiteLLM Admin UI.
|
||||||
|
|
||||||
1. Navigate to the **Agents** tab
|
1. Navigate to the **Agents** tab
|
||||||
|
|
@ -41,118 +45,32 @@ You can add A2A-compatible agents through the LiteLLM Admin UI.
|
||||||
|
|
||||||
The URL should be the invocation URL for your A2A agent (e.g., `http://localhost:10001`).
|
The URL should be the invocation URL for your A2A agent (e.g., `http://localhost:10001`).
|
||||||
|
|
||||||
|
|
||||||
|
### Add Azure AI Foundry Agents
|
||||||
|
|
||||||
|
Follow [this guide, to add your azure ai foundry agent to LiteLLM Agent Gateway](./providers/azure_ai_agents#litellm-a2a-gateway)
|
||||||
|
|
||||||
|
### Add Vertex AI Agent Engine
|
||||||
|
|
||||||
|
Follow [this guide, to add your Vertex AI Agent Engine to LiteLLM Agent Gateway](./providers/vertex_ai_agent_engine)
|
||||||
|
|
||||||
|
### Add Bedrock AgentCore Agents
|
||||||
|
|
||||||
|
Follow [this guide, to add your bedrock agentcore agent to LiteLLM Agent Gateway](./providers/bedrock_agentcore#litellm-a2a-gateway)
|
||||||
|
|
||||||
|
### Add LangGraph Agents
|
||||||
|
|
||||||
|
Follow [this guide, to add your langgraph agent to LiteLLM Agent Gateway](./providers/langgraph#litellm-a2a-gateway)
|
||||||
|
|
||||||
|
### Add Pydantic AI Agents
|
||||||
|
|
||||||
|
Follow [this guide, to add your pydantic ai agent to LiteLLM Agent Gateway](./providers/pydantic_ai_agent#litellm-a2a-gateway)
|
||||||
|
|
||||||
## Invoking your Agents
|
## Invoking your Agents
|
||||||
|
|
||||||
Use the [A2A Python SDK](https://pypi.org/project/a2a/) to invoke agents through LiteLLM.
|
See the [Invoking A2A Agents](./a2a_invoking_agents) guide to learn how to call your agents using:
|
||||||
|
- **A2A SDK** - Native A2A protocol with full support for tasks and artifacts
|
||||||
This example shows how to:
|
- **OpenAI SDK** - Familiar `/chat/completions` interface with `a2a/` model prefix
|
||||||
1. **List available agents** - Query `/v1/agents` to see which agents your key can access
|
|
||||||
2. **Select an agent** - Pick an agent from the list
|
|
||||||
3. **Invoke via A2A** - Use the A2A protocol to send messages to the agent
|
|
||||||
|
|
||||||
```python showLineNumbers title="invoke_a2a_agent.py"
|
|
||||||
from uuid import uuid4
|
|
||||||
import httpx
|
|
||||||
import asyncio
|
|
||||||
from a2a.client import A2ACardResolver, A2AClient
|
|
||||||
from a2a.types import MessageSendParams, SendMessageRequest
|
|
||||||
|
|
||||||
# === CONFIGURE THESE ===
|
|
||||||
LITELLM_BASE_URL = "http://localhost:4000" # Your LiteLLM proxy URL
|
|
||||||
LITELLM_VIRTUAL_KEY = "sk-1234" # Your LiteLLM Virtual Key
|
|
||||||
# =======================
|
|
||||||
|
|
||||||
async def main():
|
|
||||||
headers = {"Authorization": f"Bearer {LITELLM_VIRTUAL_KEY}"}
|
|
||||||
|
|
||||||
async with httpx.AsyncClient(headers=headers) as client:
|
|
||||||
# Step 1: List available agents
|
|
||||||
response = await client.get(f"{LITELLM_BASE_URL}/v1/agents")
|
|
||||||
agents = response.json()
|
|
||||||
|
|
||||||
print("Available agents:")
|
|
||||||
for agent in agents:
|
|
||||||
print(f" - {agent['agent_name']} (ID: {agent['agent_id']})")
|
|
||||||
|
|
||||||
if not agents:
|
|
||||||
print("No agents available for this key")
|
|
||||||
return
|
|
||||||
|
|
||||||
# Step 2: Select an agent and invoke it
|
|
||||||
selected_agent = agents[0]
|
|
||||||
agent_id = selected_agent["agent_id"]
|
|
||||||
agent_name = selected_agent["agent_name"]
|
|
||||||
print(f"\nInvoking: {agent_name}")
|
|
||||||
|
|
||||||
# Step 3: Use A2A protocol to invoke the agent
|
|
||||||
base_url = f"{LITELLM_BASE_URL}/a2a/{agent_id}"
|
|
||||||
resolver = A2ACardResolver(httpx_client=client, base_url=base_url)
|
|
||||||
agent_card = await resolver.get_agent_card()
|
|
||||||
a2a_client = A2AClient(httpx_client=client, agent_card=agent_card)
|
|
||||||
|
|
||||||
request = SendMessageRequest(
|
|
||||||
id=str(uuid4()),
|
|
||||||
params=MessageSendParams(
|
|
||||||
message={
|
|
||||||
"role": "user",
|
|
||||||
"parts": [{"kind": "text", "text": "Hello, what can you do?"}],
|
|
||||||
"messageId": uuid4().hex,
|
|
||||||
}
|
|
||||||
),
|
|
||||||
)
|
|
||||||
response = await a2a_client.send_message(request)
|
|
||||||
print(f"Response: {response.model_dump(mode='json', exclude_none=True, indent=4)}")
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
asyncio.run(main())
|
|
||||||
```
|
|
||||||
|
|
||||||
### Streaming Responses
|
|
||||||
|
|
||||||
For streaming responses, use `send_message_streaming`:
|
|
||||||
|
|
||||||
```python showLineNumbers title="invoke_a2a_agent_streaming.py"
|
|
||||||
from uuid import uuid4
|
|
||||||
import httpx
|
|
||||||
import asyncio
|
|
||||||
from a2a.client import A2ACardResolver, A2AClient
|
|
||||||
from a2a.types import MessageSendParams, SendStreamingMessageRequest
|
|
||||||
|
|
||||||
# === CONFIGURE THESE ===
|
|
||||||
LITELLM_BASE_URL = "http://localhost:4000" # Your LiteLLM proxy URL
|
|
||||||
LITELLM_VIRTUAL_KEY = "sk-1234" # Your LiteLLM Virtual Key
|
|
||||||
LITELLM_AGENT_NAME = "ij-local" # Agent name registered in LiteLLM
|
|
||||||
# =======================
|
|
||||||
|
|
||||||
async def main():
|
|
||||||
base_url = f"{LITELLM_BASE_URL}/a2a/{LITELLM_AGENT_NAME}"
|
|
||||||
headers = {"Authorization": f"Bearer {LITELLM_VIRTUAL_KEY}"}
|
|
||||||
|
|
||||||
async with httpx.AsyncClient(headers=headers) as httpx_client:
|
|
||||||
# Resolve agent card and create client
|
|
||||||
resolver = A2ACardResolver(httpx_client=httpx_client, base_url=base_url)
|
|
||||||
agent_card = await resolver.get_agent_card()
|
|
||||||
client = A2AClient(httpx_client=httpx_client, agent_card=agent_card)
|
|
||||||
|
|
||||||
# Send a streaming message
|
|
||||||
request = SendStreamingMessageRequest(
|
|
||||||
id=str(uuid4()),
|
|
||||||
params=MessageSendParams(
|
|
||||||
message={
|
|
||||||
"role": "user",
|
|
||||||
"parts": [{"kind": "text", "text": "Hello, what can you do?"}],
|
|
||||||
"messageId": uuid4().hex,
|
|
||||||
}
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
# Stream the response
|
|
||||||
async for chunk in client.send_message_streaming(request):
|
|
||||||
print(chunk.model_dump(mode="json", exclude_none=True))
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
asyncio.run(main())
|
|
||||||
```
|
|
||||||
|
|
||||||
## Tracking Agent Logs
|
## Tracking Agent Logs
|
||||||
|
|
||||||
|
|
@ -168,6 +86,120 @@ The logs show:
|
||||||
style={{width: '100%', display: 'block', margin: '2rem auto'}}
|
style={{width: '100%', display: 'block', margin: '2rem auto'}}
|
||||||
/>
|
/>
|
||||||
|
|
||||||
|
|
||||||
|
## Forwarding LiteLLM Context Headers
|
||||||
|
|
||||||
|
When LiteLLM invokes your A2A agent, it sends special headers that enable:
|
||||||
|
- **Trace Grouping**: All LLM calls from the same agent execution appear under one trace
|
||||||
|
- **Agent Spend Tracking**: Costs are attributed to the specific agent
|
||||||
|
|
||||||
|
| Header | Purpose |
|
||||||
|
|--------|---------|
|
||||||
|
| `X-LiteLLM-Trace-Id` | Links all LLM calls to the same execution flow |
|
||||||
|
| `X-LiteLLM-Agent-Id` | Attributes spend to the correct agent |
|
||||||
|
|
||||||
|
|
||||||
|
To enable these features, your A2A server must **forward these headers** to any LLM calls it makes back to LiteLLM.
|
||||||
|
|
||||||
|
### Implementation Steps
|
||||||
|
|
||||||
|
**Step 1: Extract headers from incoming A2A request**
|
||||||
|
```python def get_litellm_headers(request) -> dict:
|
||||||
|
"""Extract X-LiteLLM-* headers from incoming A2A request."""
|
||||||
|
all_headers = request.call_context.state.get('headers', {})
|
||||||
|
return {
|
||||||
|
k: v for k, v in all_headers.items()
|
||||||
|
if k.lower().startswith('x-litellm-')
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**Step 2: Forward headers to your LLM calls**
|
||||||
|
Pass the extracted headers when making calls back to LiteLLM:
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="openai" label="OpenAI SDK" default>
|
||||||
|
|
||||||
|
```python from openai import OpenAI
|
||||||
|
|
||||||
|
headers = get_litellm_headers(request)
|
||||||
|
|
||||||
|
client = OpenAI(
|
||||||
|
api_key="sk-your-litellm-key",
|
||||||
|
base_url="http://localhost:4000",
|
||||||
|
default_headers=headers, # Forward headers
|
||||||
|
)
|
||||||
|
|
||||||
|
response = client.chat.completions.create(
|
||||||
|
model="gpt-4o",
|
||||||
|
messages=[{"role": "user", "content": "Hello"}]
|
||||||
|
)
|
||||||
|
```
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="langchain" label="LangChain">
|
||||||
|
|
||||||
|
```python
|
||||||
|
from langchain_openai import ChatOpenAI
|
||||||
|
|
||||||
|
headers = get_litellm_headers(request)
|
||||||
|
|
||||||
|
llm = ChatOpenAI(
|
||||||
|
model="gpt-4o",
|
||||||
|
openai_api_key="sk-your-litellm-key",
|
||||||
|
base_url="http://localhost:4000",
|
||||||
|
default_headers=headers, # Forward headers
|
||||||
|
)
|
||||||
|
```
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="litellm" label="LiteLLM SDK">
|
||||||
|
|
||||||
|
```python
|
||||||
|
import litellm
|
||||||
|
|
||||||
|
headers = get_litellm_headers(request)
|
||||||
|
|
||||||
|
response = litellm.completion(
|
||||||
|
model="gpt-4o",
|
||||||
|
messages=[{"role": "user", "content": "Hello"}],
|
||||||
|
api_base="http://localhost:4000",
|
||||||
|
extra_headers=headers, # Forward headers
|
||||||
|
)
|
||||||
|
```
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="requests" label="HTTP (requests/httpx)">
|
||||||
|
|
||||||
|
```python
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
headers = get_litellm_headers(request)
|
||||||
|
headers["Authorization"] = "Bearer sk-your-litellm-key"
|
||||||
|
|
||||||
|
response = httpx.post(
|
||||||
|
"http://localhost:4000/v1/chat/completions",
|
||||||
|
headers=headers,
|
||||||
|
json={"model": "gpt-4o", "messages": [{"role": "user", "content": "Hello"}]}
|
||||||
|
)
|
||||||
|
```
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### Result
|
||||||
|
|
||||||
|
With header forwarding enabled, you'll see:
|
||||||
|
|
||||||
|
**Trace Grouping in Langfuse:**
|
||||||
|
|
||||||
|
<Image
|
||||||
|
img={require('../img/a2a_trace_grouping.png')}
|
||||||
|
style={{width: '80%', display: 'block', margin: '0', borderRadius: '8px'}}
|
||||||
|
/>
|
||||||
|
|
||||||
|
**Agent Spend Attribution:**
|
||||||
|
|
||||||
|
<Image
|
||||||
|
img={require('../img/a2a_agent_spend.png')}
|
||||||
|
style={{width: '80%', display: 'block', margin: '0', borderRadius: '8px'}}
|
||||||
|
/>
|
||||||
|
|
||||||
## API Reference
|
## API Reference
|
||||||
|
|
||||||
### Endpoint
|
### Endpoint
|
||||||
|
|
|
||||||
147
docs/my-website/docs/a2a_cost_tracking.md
Normal file
147
docs/my-website/docs/a2a_cost_tracking.md
Normal file
|
|
@ -0,0 +1,147 @@
|
||||||
|
import Image from '@theme/IdealImage';
|
||||||
|
|
||||||
|
# A2A Agent Cost Tracking
|
||||||
|
|
||||||
|
LiteLLM supports adding custom cost tracking for A2A agents. You can configure:
|
||||||
|
|
||||||
|
- **Flat cost per query** - A fixed cost charged for each agent request
|
||||||
|
- **Cost by input/output tokens** - Variable cost based on token usage
|
||||||
|
|
||||||
|
This allows you to track and attribute costs for agent usage across your organization, making it easy to see how much each team or project is spending on agent calls.
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
### 1. Navigate to Agents
|
||||||
|
|
||||||
|
From the sidebar, click on "Agents" to open the agent management page.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 2. Create a New Agent
|
||||||
|
|
||||||
|
Click "+ Add New Agent" to open the creation form. You'll need to provide a few basic details:
|
||||||
|
|
||||||
|
- **Agent Name** - A unique identifier for your agent (used in API calls)
|
||||||
|
- **Display Name** - A human-readable name shown in the UI
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 3. Configure Cost Settings
|
||||||
|
|
||||||
|
Scroll down and click on "Cost Configuration" to expand the cost settings panel. This is where you define how much to charge for agent usage.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 4. Set Cost Per Query
|
||||||
|
|
||||||
|
Enter the cost per query amount (in dollars). For example, entering `0.05` means each request to this agent will be charged $0.05.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 5. Create the Agent
|
||||||
|
|
||||||
|
Once you've configured everything, click "Create Agent" to save. Your agent is now ready to use with cost tracking enabled.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
## Testing Cost Tracking
|
||||||
|
|
||||||
|
Let's verify that cost tracking is working by sending a test request through the Playground.
|
||||||
|
|
||||||
|
### 1. Go to Playground
|
||||||
|
|
||||||
|
Click "Playground" in the sidebar to open the interactive testing interface.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 2. Select A2A Endpoint
|
||||||
|
|
||||||
|
By default, the Playground uses the chat completions endpoint. To test your agent, click "Endpoint Type" and select `/v1/a2a/message/send` from the dropdown.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 3. Select Your Agent
|
||||||
|
|
||||||
|
Now pick the agent you just created from the agent dropdown. You should see it listed by its display name.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 4. Send a Test Message
|
||||||
|
|
||||||
|
Type a message and hit send. You can use the suggested prompts or write your own.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
Once the agent responds, the request is logged with the cost you configured.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
## Viewing Cost in Logs
|
||||||
|
|
||||||
|
Now let's confirm the cost was actually tracked.
|
||||||
|
|
||||||
|
### 1. Navigate to Logs
|
||||||
|
|
||||||
|
Click "Logs" in the sidebar to see all recent requests.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
### 2. View Cost Attribution
|
||||||
|
|
||||||
|
Find your agent request in the list. You'll see the cost column showing the amount you configured. This cost is now attributed to the API key that made the request, so you can track spend per team or project.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
## View Spend in Usage Page
|
||||||
|
|
||||||
|
Navigate to the Agent Usage tab in the Admin UI to view agent-level spend analytics:
|
||||||
|
|
||||||
|
### 1. Access Agent Usage
|
||||||
|
|
||||||
|
Go to the Usage page in the Admin UI (`PROXY_BASE_URL/ui/?login=success&page=new_usage`) and click on the **Agent Usage** tab.
|
||||||
|
|
||||||
|
<Image img={require('../img/agent_usage_ui_navigation.png')} />
|
||||||
|
|
||||||
|
### 2. View Agent Analytics
|
||||||
|
|
||||||
|
The Agent Usage dashboard provides:
|
||||||
|
|
||||||
|
- **Total spend per agent**: View aggregated spend across all agents
|
||||||
|
- **Daily spend trends**: See how agent spend changes over time
|
||||||
|
- **Model usage breakdown**: Understand which models each agent uses
|
||||||
|
- **Activity metrics**: Track requests, tokens, and success rates per agent
|
||||||
|
|
||||||
|
<Image img={require('../img/agent_usage_analytics.png')} />
|
||||||
|
|
||||||
|
### 3. Filter by Agent
|
||||||
|
|
||||||
|
Use the agent filter dropdown to view spend for specific agents:
|
||||||
|
|
||||||
|
- Select one or more agent IDs from the dropdown
|
||||||
|
- View filtered analytics, spend logs, and activity metrics
|
||||||
|
- Compare spend across different agents
|
||||||
|
|
||||||
|
<Image img={require('../img/agent_usage_filter.png')} />
|
||||||
|
|
||||||
|
## Cost Configuration Options
|
||||||
|
|
||||||
|
You can mix and match these options depending on your pricing model:
|
||||||
|
|
||||||
|
| Field | Description |
|
||||||
|
| ----------------------------- | ----------------------------------------- |
|
||||||
|
| **Cost Per Query ($)** | Fixed cost charged for each agent request |
|
||||||
|
| **Input Cost Per Token ($)** | Cost per input token processed |
|
||||||
|
| **Output Cost Per Token ($)** | Cost per output token generated |
|
||||||
|
|
||||||
|
For most use cases, a flat cost per query is simplest. Use token-based pricing if your agent costs vary significantly based on input/output length.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [A2A Agent Gateway](./a2a.md)
|
||||||
|
- [Spend Tracking](./proxy/cost_tracking.md)
|
||||||
280
docs/my-website/docs/a2a_invoking_agents.md
Normal file
280
docs/my-website/docs/a2a_invoking_agents.md
Normal file
|
|
@ -0,0 +1,280 @@
|
||||||
|
import Tabs from '@theme/Tabs';
|
||||||
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
# Invoking A2A Agents
|
||||||
|
|
||||||
|
Learn how to invoke A2A agents through LiteLLM using different methods.
|
||||||
|
|
||||||
|
:::tip Deploy Your Own A2A Agent
|
||||||
|
|
||||||
|
Want to test with your own agent? Deploy this template A2A agent powered by Google Gemini:
|
||||||
|
|
||||||
|
[**shin-bot-litellm/a2a-gemini-agent**](https://github.com/shin-bot-litellm/a2a-gemini-agent) - Simple deployable A2A agent with streaming support
|
||||||
|
|
||||||
|
:::
|
||||||
|
|
||||||
|
## A2A SDK
|
||||||
|
|
||||||
|
Use the [A2A Python SDK](https://pypi.org/project/a2a-sdk) to invoke agents through LiteLLM using the A2A protocol.
|
||||||
|
|
||||||
|
### Non-Streaming
|
||||||
|
|
||||||
|
This example shows how to:
|
||||||
|
1. **List available agents** - Query `/v1/agents` to see which agents your key can access
|
||||||
|
2. **Select an agent** - Pick an agent from the list
|
||||||
|
3. **Invoke via A2A** - Use the A2A protocol to send messages to the agent
|
||||||
|
|
||||||
|
```python showLineNumbers title="invoke_a2a_agent.py"
|
||||||
|
from uuid import uuid4
|
||||||
|
import httpx
|
||||||
|
import asyncio
|
||||||
|
from a2a.client import A2ACardResolver, A2AClient
|
||||||
|
from a2a.types import MessageSendParams, SendMessageRequest
|
||||||
|
|
||||||
|
# === CONFIGURE THESE ===
|
||||||
|
LITELLM_BASE_URL = "http://localhost:4000" # Your LiteLLM proxy URL
|
||||||
|
LITELLM_VIRTUAL_KEY = "sk-1234" # Your LiteLLM Virtual Key
|
||||||
|
# =======================
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
headers = {"Authorization": f"Bearer {LITELLM_VIRTUAL_KEY}"}
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(headers=headers) as client:
|
||||||
|
# Step 1: List available agents
|
||||||
|
response = await client.get(f"{LITELLM_BASE_URL}/v1/agents")
|
||||||
|
agents = response.json()
|
||||||
|
|
||||||
|
print("Available agents:")
|
||||||
|
for agent in agents:
|
||||||
|
print(f" - {agent['agent_name']} (ID: {agent['agent_id']})")
|
||||||
|
|
||||||
|
if not agents:
|
||||||
|
print("No agents available for this key")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Step 2: Select an agent and invoke it
|
||||||
|
selected_agent = agents[0]
|
||||||
|
agent_id = selected_agent["agent_id"]
|
||||||
|
agent_name = selected_agent["agent_name"]
|
||||||
|
print(f"\nInvoking: {agent_name}")
|
||||||
|
|
||||||
|
# Step 3: Use A2A protocol to invoke the agent
|
||||||
|
base_url = f"{LITELLM_BASE_URL}/a2a/{agent_id}"
|
||||||
|
resolver = A2ACardResolver(httpx_client=client, base_url=base_url)
|
||||||
|
agent_card = await resolver.get_agent_card()
|
||||||
|
a2a_client = A2AClient(httpx_client=client, agent_card=agent_card)
|
||||||
|
|
||||||
|
request = SendMessageRequest(
|
||||||
|
id=str(uuid4()),
|
||||||
|
params=MessageSendParams(
|
||||||
|
message={
|
||||||
|
"role": "user",
|
||||||
|
"parts": [{"kind": "text", "text": "Hello, what can you do?"}],
|
||||||
|
"messageId": uuid4().hex,
|
||||||
|
}
|
||||||
|
),
|
||||||
|
)
|
||||||
|
response = await a2a_client.send_message(request)
|
||||||
|
print(f"Response: {response.model_dump(mode='json', exclude_none=True, indent=4)}")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
|
```
|
||||||
|
|
||||||
|
### Streaming
|
||||||
|
|
||||||
|
For streaming responses, use `send_message_streaming`:
|
||||||
|
|
||||||
|
```python showLineNumbers title="invoke_a2a_agent_streaming.py"
|
||||||
|
from uuid import uuid4
|
||||||
|
import httpx
|
||||||
|
import asyncio
|
||||||
|
from a2a.client import A2ACardResolver, A2AClient
|
||||||
|
from a2a.types import MessageSendParams, SendStreamingMessageRequest
|
||||||
|
|
||||||
|
# === CONFIGURE THESE ===
|
||||||
|
LITELLM_BASE_URL = "http://localhost:4000" # Your LiteLLM proxy URL
|
||||||
|
LITELLM_VIRTUAL_KEY = "sk-1234" # Your LiteLLM Virtual Key
|
||||||
|
LITELLM_AGENT_NAME = "ij-local" # Agent name registered in LiteLLM
|
||||||
|
# =======================
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
base_url = f"{LITELLM_BASE_URL}/a2a/{LITELLM_AGENT_NAME}"
|
||||||
|
headers = {"Authorization": f"Bearer {LITELLM_VIRTUAL_KEY}"}
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(headers=headers) as httpx_client:
|
||||||
|
# Resolve agent card and create client
|
||||||
|
resolver = A2ACardResolver(httpx_client=httpx_client, base_url=base_url)
|
||||||
|
agent_card = await resolver.get_agent_card()
|
||||||
|
client = A2AClient(httpx_client=httpx_client, agent_card=agent_card)
|
||||||
|
|
||||||
|
# Send a streaming message
|
||||||
|
request = SendStreamingMessageRequest(
|
||||||
|
id=str(uuid4()),
|
||||||
|
params=MessageSendParams(
|
||||||
|
message={
|
||||||
|
"role": "user",
|
||||||
|
"parts": [{"kind": "text", "text": "Tell me a long story"}],
|
||||||
|
"messageId": uuid4().hex,
|
||||||
|
}
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Stream the response
|
||||||
|
async for chunk in client.send_message_streaming(request):
|
||||||
|
print(chunk.model_dump(mode="json", exclude_none=True))
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
|
```
|
||||||
|
|
||||||
|
## /chat/completions API (OpenAI SDK)
|
||||||
|
|
||||||
|
You can also invoke A2A agents using the familiar OpenAI SDK by using the `a2a/` model prefix.
|
||||||
|
|
||||||
|
### Non-Streaming
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="python" label="Python" default>
|
||||||
|
|
||||||
|
```python showLineNumbers title="openai_non_streaming.py"
|
||||||
|
import openai
|
||||||
|
|
||||||
|
client = openai.OpenAI(
|
||||||
|
api_key="sk-1234", # Your LiteLLM Virtual Key
|
||||||
|
base_url="http://localhost:4000" # Your LiteLLM proxy URL
|
||||||
|
)
|
||||||
|
|
||||||
|
response = client.chat.completions.create(
|
||||||
|
model="a2a/my-agent", # Use a2a/ prefix with your agent name
|
||||||
|
messages=[
|
||||||
|
{"role": "user", "content": "Hello, what can you do?"}
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
print(response.choices[0].message.content)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="typescript" label="TypeScript">
|
||||||
|
|
||||||
|
```typescript showLineNumbers title="openai_non_streaming.ts"
|
||||||
|
import OpenAI from 'openai';
|
||||||
|
|
||||||
|
const client = new OpenAI({
|
||||||
|
apiKey: 'sk-1234', // Your LiteLLM Virtual Key
|
||||||
|
baseURL: 'http://localhost:4000' // Your LiteLLM proxy URL
|
||||||
|
});
|
||||||
|
|
||||||
|
const response = await client.chat.completions.create({
|
||||||
|
model: 'a2a/my-agent', // Use a2a/ prefix with your agent name
|
||||||
|
messages: [
|
||||||
|
{ role: 'user', content: 'Hello, what can you do?' }
|
||||||
|
]
|
||||||
|
});
|
||||||
|
|
||||||
|
console.log(response.choices[0].message.content);
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="curl" label="cURL">
|
||||||
|
|
||||||
|
```bash showLineNumbers title="curl_non_streaming.sh"
|
||||||
|
curl -X POST http://localhost:4000/v1/chat/completions \
|
||||||
|
-H "Authorization: Bearer sk-1234" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "a2a/my-agent",
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Hello, what can you do?"}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### Streaming
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="python" label="Python" default>
|
||||||
|
|
||||||
|
```python showLineNumbers title="openai_streaming.py"
|
||||||
|
import openai
|
||||||
|
|
||||||
|
client = openai.OpenAI(
|
||||||
|
api_key="sk-1234", # Your LiteLLM Virtual Key
|
||||||
|
base_url="http://localhost:4000" # Your LiteLLM proxy URL
|
||||||
|
)
|
||||||
|
|
||||||
|
stream = client.chat.completions.create(
|
||||||
|
model="a2a/my-agent", # Use a2a/ prefix with your agent name
|
||||||
|
messages=[
|
||||||
|
{"role": "user", "content": "Tell me a long story"}
|
||||||
|
],
|
||||||
|
stream=True
|
||||||
|
)
|
||||||
|
|
||||||
|
for chunk in stream:
|
||||||
|
if chunk.choices[0].delta.content:
|
||||||
|
print(chunk.choices[0].delta.content, end="", flush=True)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="typescript" label="TypeScript">
|
||||||
|
|
||||||
|
```typescript showLineNumbers title="openai_streaming.ts"
|
||||||
|
import OpenAI from 'openai';
|
||||||
|
|
||||||
|
const client = new OpenAI({
|
||||||
|
apiKey: 'sk-1234', // Your LiteLLM Virtual Key
|
||||||
|
baseURL: 'http://localhost:4000' // Your LiteLLM proxy URL
|
||||||
|
});
|
||||||
|
|
||||||
|
const stream = await client.chat.completions.create({
|
||||||
|
model: 'a2a/my-agent', // Use a2a/ prefix with your agent name
|
||||||
|
messages: [
|
||||||
|
{ role: 'user', content: 'Tell me a long story' }
|
||||||
|
],
|
||||||
|
stream: true
|
||||||
|
});
|
||||||
|
|
||||||
|
for await (const chunk of stream) {
|
||||||
|
const content = chunk.choices[0]?.delta?.content;
|
||||||
|
if (content) {
|
||||||
|
process.stdout.write(content);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="curl" label="cURL">
|
||||||
|
|
||||||
|
```bash showLineNumbers title="curl_streaming.sh"
|
||||||
|
curl -X POST http://localhost:4000/v1/chat/completions \
|
||||||
|
-H "Authorization: Bearer sk-1234" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "a2a/my-agent",
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Tell me a long story"}
|
||||||
|
],
|
||||||
|
"stream": true
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Key Differences
|
||||||
|
|
||||||
|
| Method | Use Case | Advantages |
|
||||||
|
|--------|----------|------------|
|
||||||
|
| **A2A SDK** | Native A2A protocol integration | • Full A2A protocol support<br/>• Access to task states and artifacts<br/>• Context management |
|
||||||
|
| **OpenAI SDK** | Familiar OpenAI-style interface | • Drop-in replacement for OpenAI calls<br/>• Easier migration from LLM to agent workflows<br/>• Works with existing OpenAI tooling |
|
||||||
|
|
||||||
|
:::tip Model Prefix
|
||||||
|
|
||||||
|
When using the OpenAI SDK, always prefix your agent name with `a2a/` (e.g., `a2a/my-agent`) to route requests to the A2A agent instead of an LLM provider.
|
||||||
|
|
||||||
|
:::
|
||||||
|
|
@ -93,6 +93,12 @@ Implement `POST /beta/litellm_basic_guardrail_api`
|
||||||
"user_api_key_end_user_id": "end user id associated with the litellm virtual key used",
|
"user_api_key_end_user_id": "end user id associated with the litellm virtual key used",
|
||||||
"user_api_key_org_id": "org id associated with the litellm virtual key used"
|
"user_api_key_org_id": "org id associated with the litellm virtual key used"
|
||||||
},
|
},
|
||||||
|
"request_headers": { // optional: inbound request headers (allowlist). Allowed headers show their value; all others show "[present]" to indicate the header existed.
|
||||||
|
"User-Agent": "OpenAI/Python 2.17.0",
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"X-Request-Id": "[present]"
|
||||||
|
},
|
||||||
|
"litellm_version": "1.x.y", // optional: LiteLLM library version running this proxy
|
||||||
"input_type": "request", // "request" or "response"
|
"input_type": "request", // "request" or "response"
|
||||||
"litellm_call_id": "unique_call_id", // the call id of the individual LLM call
|
"litellm_call_id": "unique_call_id", // the call id of the individual LLM call
|
||||||
"litellm_trace_id": "trace_id", // the trace id of the LLM call - useful if there are multiple LLM calls for the same conversation
|
"litellm_trace_id": "trace_id", // the trace id of the LLM call - useful if there are multiple LLM calls for the same conversation
|
||||||
|
|
@ -237,6 +243,27 @@ litellm_settings:
|
||||||
language: "en"
|
language: "en"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Example: Pillar Security
|
||||||
|
|
||||||
|
[Pillar Security](https://pillar.security) uses the Generic Guardrail API to provide comprehensive AI security scanning including prompt injection protection, PII/PCI detection, secret detection, and content moderation.
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
guardrails:
|
||||||
|
- guardrail_name: "pillar-security"
|
||||||
|
litellm_params:
|
||||||
|
guardrail: generic_guardrail_api
|
||||||
|
mode: [pre_call, post_call]
|
||||||
|
api_base: https://api.pillar.security/api/v1/integrations/litellm
|
||||||
|
api_key: os.environ/PILLAR_API_KEY
|
||||||
|
default_on: true
|
||||||
|
additional_provider_specific_params:
|
||||||
|
plr_mask: true # Enable automatic masking of sensitive data
|
||||||
|
plr_evidence: true # Include detection evidence in response
|
||||||
|
plr_scanners: true # Include scanner details in response
|
||||||
|
```
|
||||||
|
|
||||||
|
See the [Pillar Security documentation](../proxy/guardrails/pillar_security.md) for full configuration options.
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
Users apply your guardrail by name:
|
Users apply your guardrail by name:
|
||||||
|
|
|
||||||
|
|
@ -101,12 +101,11 @@ model_list:
|
||||||
- model_name: gpt-4
|
- model_name: gpt-4
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: gpt-4
|
model: gpt-4
|
||||||
api_key: os.environ/OPENAI_API_KEY
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
|
||||||
litellm_settings:
|
guardrails:
|
||||||
guardrails:
|
|
||||||
- guardrail_name: my_guardrail
|
- guardrail_name: my_guardrail
|
||||||
litellm_params:
|
litellm_params:
|
||||||
guardrail: my_guardrail
|
guardrail: my_guardrail
|
||||||
mode: during_call
|
mode: during_call
|
||||||
api_key: os.environ/MY_GUARDRAIL_API_KEY
|
api_key: os.environ/MY_GUARDRAIL_API_KEY
|
||||||
|
|
|
||||||
|
|
@ -92,6 +92,7 @@ model_list:
|
||||||
model: vertex_ai/claude-3-5-sonnet-v2@20241022
|
model: vertex_ai/claude-3-5-sonnet-v2@20241022
|
||||||
vertex_project: my-project
|
vertex_project: my-project
|
||||||
vertex_location: us-east5
|
vertex_location: us-east5
|
||||||
|
vertex_count_tokens_location: us-east5 # Optional: Override location for token counting (count_tokens not available on global location)
|
||||||
|
|
||||||
- model_name: claude-bedrock
|
- model_name: claude-bedrock
|
||||||
litellm_params:
|
litellm_params:
|
||||||
|
|
|
||||||
294
docs/my-website/docs/anthropic_unified/structured_output.md
Normal file
294
docs/my-website/docs/anthropic_unified/structured_output.md
Normal file
|
|
@ -0,0 +1,294 @@
|
||||||
|
import Tabs from '@theme/Tabs';
|
||||||
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
# Structured Output /v1/messages
|
||||||
|
|
||||||
|
Use LiteLLM to call Anthropic's structured output feature via the `/v1/messages` endpoint.
|
||||||
|
|
||||||
|
## Supported Providers
|
||||||
|
|
||||||
|
| Provider | Supported | Notes |
|
||||||
|
|----------|-----------|-------|
|
||||||
|
| Anthropic | ✅ | Native support |
|
||||||
|
| Azure AI (Anthropic models) | ✅ | Claude models on Azure AI |
|
||||||
|
| Bedrock (Converse Anthropic models) | ✅ | Claude models via Bedrock Converse API |
|
||||||
|
| Bedrock (Invoke Anthropic models) | ✅ | Claude models via Bedrock Invoke API |
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
### LiteLLM Proxy Server
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="anthropic" label="Anthropic">
|
||||||
|
|
||||||
|
1. Setup config.yaml
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: claude-sonnet
|
||||||
|
litellm_params:
|
||||||
|
model: anthropic/claude-sonnet-4-5-20250514
|
||||||
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Start proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Test it!
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://localhost:4000/v1/messages \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||||
|
-H "anthropic-version: 2023-06-01" \
|
||||||
|
-d '{
|
||||||
|
"model": "claude-sonnet",
|
||||||
|
"max_tokens": 1024,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Extract the key information from this email: John Smith (john@example.com) is interested in our Enterprise plan and wants to schedule a demo for next Tuesday at 2pm."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"email": {"type": "string"},
|
||||||
|
"plan_interest": {"type": "string"},
|
||||||
|
"demo_requested": {"type": "boolean"}
|
||||||
|
},
|
||||||
|
"required": ["name", "email", "plan_interest", "demo_requested"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="azure_ai" label="Azure AI (Anthropic)">
|
||||||
|
|
||||||
|
1. Setup config.yaml
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: azure-claude-sonnet
|
||||||
|
litellm_params:
|
||||||
|
model: azure_ai/claude-sonnet-4-5-20250514
|
||||||
|
api_key: os.environ/AZURE_AI_API_KEY
|
||||||
|
api_base: https://your-endpoint.inference.ai.azure.com
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Start proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Test it!
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://localhost:4000/v1/messages \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||||
|
-H "anthropic-version: 2023-06-01" \
|
||||||
|
-d '{
|
||||||
|
"model": "azure-claude-sonnet",
|
||||||
|
"max_tokens": 1024,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Extract the key information from this email: John Smith (john@example.com) is interested in our Enterprise plan and wants to schedule a demo for next Tuesday at 2pm."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"email": {"type": "string"},
|
||||||
|
"plan_interest": {"type": "string"},
|
||||||
|
"demo_requested": {"type": "boolean"}
|
||||||
|
},
|
||||||
|
"required": ["name", "email", "plan_interest", "demo_requested"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="bedrock" label="Bedrock (Converse)">
|
||||||
|
|
||||||
|
1. Setup config.yaml
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: bedrock-claude-sonnet
|
||||||
|
litellm_params:
|
||||||
|
model: bedrock/global.anthropic.claude-sonnet-4-5-20250929-v1:0
|
||||||
|
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||||
|
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||||
|
aws_region_name: us-west-2
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Start proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Test it!
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://localhost:4000/v1/messages \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||||
|
-H "anthropic-version: 2023-06-01" \
|
||||||
|
-d '{
|
||||||
|
"model": "bedrock-claude-sonnet",
|
||||||
|
"max_tokens": 1024,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Extract the key information from this email: John Smith (john@example.com) is interested in our Enterprise plan and wants to schedule a demo for next Tuesday at 2pm."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"email": {"type": "string"},
|
||||||
|
"plan_interest": {"type": "string"},
|
||||||
|
"demo_requested": {"type": "boolean"}
|
||||||
|
},
|
||||||
|
"required": ["name", "email", "plan_interest", "demo_requested"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
|
||||||
|
<TabItem value="bedrock_invoke" label="Bedrock (Invoke)">
|
||||||
|
|
||||||
|
1. Setup config.yaml
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model_list:
|
||||||
|
- model_name: bedrock-claude-invoke
|
||||||
|
litellm_params:
|
||||||
|
model: bedrock/invoke/global.anthropic.claude-sonnet-4-5-20250929-v1:0
|
||||||
|
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||||
|
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||||
|
aws_region_name: us-west-2
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Start proxy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
litellm --config /path/to/config.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Test it!
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://localhost:4000/v1/messages \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||||
|
-H "anthropic-version: 2023-06-01" \
|
||||||
|
-d '{
|
||||||
|
"model": "bedrock-claude-invoke",
|
||||||
|
"max_tokens": 1024,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Extract the key information from this email: John Smith (john@example.com) is interested in our Enterprise plan and wants to schedule a demo for next Tuesday at 2pm."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"output_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"email": {"type": "string"},
|
||||||
|
"plan_interest": {"type": "string"},
|
||||||
|
"demo_requested": {"type": "boolean"}
|
||||||
|
},
|
||||||
|
"required": ["name", "email", "plan_interest", "demo_requested"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
## Example Response
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"id": "msg_01XFDUDYJgAACzvnptvVoYEL",
|
||||||
|
"type": "message",
|
||||||
|
"role": "assistant",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"text": "{\"name\":\"John Smith\",\"email\":\"john@example.com\",\"plan_interest\":\"Enterprise\",\"demo_requested\":true}"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"model": "claude-sonnet-4-5-20250514",
|
||||||
|
"stop_reason": "end_turn",
|
||||||
|
"stop_sequence": null,
|
||||||
|
"usage": {
|
||||||
|
"input_tokens": 75,
|
||||||
|
"output_tokens": 28
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Request Format
|
||||||
|
|
||||||
|
### output_format
|
||||||
|
|
||||||
|
The `output_format` parameter specifies the structured output format.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"output_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"field_name": {"type": "string"},
|
||||||
|
"another_field": {"type": "integer"}
|
||||||
|
},
|
||||||
|
"required": ["field_name", "another_field"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Fields
|
||||||
|
|
||||||
|
- **type** (string): Must be `"json_schema"`
|
||||||
|
- **schema** (object): A JSON Schema object defining the expected output structure
|
||||||
|
- **type** (string): The root type, typically `"object"`
|
||||||
|
- **properties** (object): Defines the fields and their types
|
||||||
|
- **required** (array): List of required field names
|
||||||
|
- **additionalProperties** (boolean): Set to `false` to enforce strict schema adherence
|
||||||
|
|
@ -7,7 +7,7 @@ Covers Batches, Files
|
||||||
|
|
||||||
| Feature | Supported | Notes |
|
| Feature | Supported | Notes |
|
||||||
|-------|-------|-------|
|
|-------|-------|-------|
|
||||||
| Supported Providers | OpenAI, Azure, Vertex, Bedrock | - |
|
| Supported Providers | OpenAI, Azure, Vertex, Bedrock, vLLM | - |
|
||||||
| ✨ Cost Tracking | ✅ | LiteLLM Enterprise only |
|
| ✨ Cost Tracking | ✅ | LiteLLM Enterprise only |
|
||||||
| Logging | ✅ | Works across all logging integrations |
|
| Logging | ✅ | Works across all logging integrations |
|
||||||
|
|
||||||
|
|
@ -430,6 +430,7 @@ All batch and file endpoints support model-based routing:
|
||||||
### [OpenAI](#quick-start)
|
### [OpenAI](#quick-start)
|
||||||
### [Vertex AI](./providers/vertex#batch-apis)
|
### [Vertex AI](./providers/vertex#batch-apis)
|
||||||
### [Bedrock](./providers/bedrock_batches)
|
### [Bedrock](./providers/bedrock_batches)
|
||||||
|
### [vLLM](./providers/vllm_batches)
|
||||||
|
|
||||||
|
|
||||||
## How Cost Tracking for Batches API Works
|
## How Cost Tracking for Batches API Works
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,13 @@ import Image from '@theme/IdealImage';
|
||||||
|
|
||||||
Benchmarks for LiteLLM Gateway (Proxy Server) tested against a fake OpenAI endpoint.
|
Benchmarks for LiteLLM Gateway (Proxy Server) tested against a fake OpenAI endpoint.
|
||||||
|
|
||||||
|
## Setting Up a Fake OpenAI Endpoint
|
||||||
|
|
||||||
|
For load testing and benchmarking, you can use a fake OpenAI proxy server. LiteLLM provides:
|
||||||
|
|
||||||
|
1. **Hosted endpoint**: Use our free hosted fake endpoint at `https://exampleopenaiendpoint-production.up.railway.app/`
|
||||||
|
2. **Self-hosted**: Set up your own fake OpenAI proxy server using [github.com/BerriAI/example_openai_endpoint](https://github.com/BerriAI/example_openai_endpoint)
|
||||||
|
|
||||||
Use this config for testing:
|
Use this config for testing:
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
|
|
@ -12,7 +19,7 @@ model_list:
|
||||||
- model_name: "fake-openai-endpoint"
|
- model_name: "fake-openai-endpoint"
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: openai/any
|
model: openai/any
|
||||||
api_base: https://your-fake-openai-endpoint.com/chat/completions
|
api_base: https://exampleopenaiendpoint-production.up.railway.app/ # or your self-hosted endpoint
|
||||||
api_key: "test"
|
api_key: "test"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -48,6 +55,28 @@ In these tests the baseline latency characteristics are measured against a fake-
|
||||||
- High-percentile latencies drop significantly: P95 630 ms → 150 ms, P99 1,200 ms → 240 ms.
|
- High-percentile latencies drop significantly: P95 630 ms → 150 ms, P99 1,200 ms → 240 ms.
|
||||||
- Setting workers equal to CPU count gives optimal performance.
|
- Setting workers equal to CPU count gives optimal performance.
|
||||||
|
|
||||||
|
## `/realtime` API Benchmarks
|
||||||
|
|
||||||
|
End-to-end latency benchmarks for the `/realtime` endpoint tested against a fake realtime endpoint.
|
||||||
|
|
||||||
|
### Performance Metrics
|
||||||
|
|
||||||
|
| Metric | Value |
|
||||||
|
| --------------- | ---------- |
|
||||||
|
| Median latency | 59 ms |
|
||||||
|
| p95 latency | 67 ms |
|
||||||
|
| p99 latency | 99 ms |
|
||||||
|
| Average latency | 63 ms |
|
||||||
|
| RPS | 1,207 |
|
||||||
|
|
||||||
|
### Test Setup
|
||||||
|
|
||||||
|
| Category | Specification |
|
||||||
|
|----------|---------------|
|
||||||
|
| **Load Testing** | Locust: 1,000 concurrent users, 500 ramp-up |
|
||||||
|
| **System** | 4 vCPUs, 8 GB RAM, 4 workers, 4 instances |
|
||||||
|
| **Database** | PostgreSQL (Redis unused) |
|
||||||
|
|
||||||
## Machine Spec used for testing
|
## Machine Spec used for testing
|
||||||
|
|
||||||
Each machine deploying LiteLLM had the following specs:
|
Each machine deploying LiteLLM had the following specs:
|
||||||
|
|
@ -60,6 +89,58 @@ Each machine deploying LiteLLM had the following specs:
|
||||||
- Database: PostgreSQL
|
- Database: PostgreSQL
|
||||||
- Redis: Not used
|
- Redis: Not used
|
||||||
|
|
||||||
|
## Infrastructure Recommendations
|
||||||
|
|
||||||
|
Recommended specifications based on benchmark results and industry standards for API gateway deployments.
|
||||||
|
|
||||||
|
### PostgreSQL
|
||||||
|
|
||||||
|
Required for authentication, key management, and usage tracking.
|
||||||
|
|
||||||
|
| Workload | CPU | RAM | Storage | Connections |
|
||||||
|
|----------|-----|-----|---------|-------------|
|
||||||
|
| 1-2K RPS | 4-8 cores | 16GB | 200GB SSD (3000+ IOPS) | 100-200 |
|
||||||
|
| 2-5K RPS | 8 cores | 16-32GB | 500GB SSD (5000+ IOPS) | 200-500 |
|
||||||
|
| 5K+ RPS | 16+ cores | 32-64GB | 1TB+ SSD (10000+ IOPS) | 500+ |
|
||||||
|
|
||||||
|
**Configuration:** Set `proxy_batch_write_at: 60` to batch writes and reduce DB load. Total connections = pool limit × instances.
|
||||||
|
|
||||||
|
### Redis (Recommended)
|
||||||
|
|
||||||
|
Redis was not used in these benchmarks but provides significant production benefits: 60-80% reduced DB load.
|
||||||
|
|
||||||
|
| Workload | CPU | RAM |
|
||||||
|
|----------|-----|-----|
|
||||||
|
| 1-2K RPS | 2-4 cores | 8GB |
|
||||||
|
| 2-5K RPS | 4 cores | 16GB |
|
||||||
|
| 5K+ RPS | 8+ cores | 32GB+ |
|
||||||
|
|
||||||
|
**Requirements:** Redis 7.0+, AOF persistence enabled, `allkeys-lru` eviction policy.
|
||||||
|
|
||||||
|
**Configuration:**
|
||||||
|
```yaml
|
||||||
|
router_settings:
|
||||||
|
redis_host: os.environ/REDIS_HOST
|
||||||
|
redis_port: os.environ/REDIS_PORT
|
||||||
|
redis_password: os.environ/REDIS_PASSWORD
|
||||||
|
|
||||||
|
litellm_settings:
|
||||||
|
cache: True
|
||||||
|
cache_params:
|
||||||
|
type: redis
|
||||||
|
host: os.environ/REDIS_HOST
|
||||||
|
port: os.environ/REDIS_PORT
|
||||||
|
password: os.environ/REDIS_PASSWORD
|
||||||
|
```
|
||||||
|
|
||||||
|
:::tip
|
||||||
|
Use `redis_host`, `redis_port`, and `redis_password` instead of `redis_url` for ~80 RPS better performance.
|
||||||
|
:::
|
||||||
|
|
||||||
|
**Scaling:** DB connections scale linearly with instances. Consider PostgreSQL read replicas beyond 5K RPS.
|
||||||
|
|
||||||
|
See [Production Configuration](./proxy/prod) for detailed best practices.
|
||||||
|
|
||||||
## Locust Settings
|
## Locust Settings
|
||||||
|
|
||||||
- 1000 Users
|
- 1000 Users
|
||||||
|
|
@ -172,7 +253,7 @@ class MyUser(HttpUser):
|
||||||
|
|
||||||
## Logging Callbacks
|
## Logging Callbacks
|
||||||
|
|
||||||
### [GCS Bucket Logging](https://docs.litellm.ai/docs/proxy/bucket)
|
### [GCS Bucket Logging](https://docs.litellm.ai/docs/observability/gcs_bucket_integration)
|
||||||
|
|
||||||
Using GCS Bucket has **no impact on latency, RPS compared to Basic Litellm Proxy**
|
Using GCS Bucket has **no impact on latency, RPS compared to Basic Litellm Proxy**
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -105,6 +105,14 @@ Then simply initialize:
|
||||||
litellm.cache = Cache(type="redis")
|
litellm.cache = Cache(type="redis")
|
||||||
```
|
```
|
||||||
|
|
||||||
|
:::info
|
||||||
|
Use `REDIS_*` environment variables as the primary mechanism for configuring all Redis client library parameters. This approach automatically maps environment variables to Redis client kwargs and is the suggested way to toggle Redis settings.
|
||||||
|
:::
|
||||||
|
|
||||||
|
:::warning
|
||||||
|
If you need to pass non-string Redis parameters (integers, booleans, complex objects), avoid `REDIS_*` environment variables as they may fail during Redis client initialization. Instead, pass them directly as kwargs to the `Cache()` constructor.
|
||||||
|
:::
|
||||||
|
|
||||||
</TabItem>
|
</TabItem>
|
||||||
|
|
||||||
<TabItem value="gcs" label="gcs-cache">
|
<TabItem value="gcs" label="gcs-cache">
|
||||||
|
|
|
||||||
|
|
@ -142,7 +142,47 @@ def completion(
|
||||||
- `tool_call_id`: *str (optional)* - Tool call that this message is responding to.
|
- `tool_call_id`: *str (optional)* - Tool call that this message is responding to.
|
||||||
|
|
||||||
|
|
||||||
[**See All Message Values**](https://github.com/BerriAI/litellm/blob/8600ec77042dacad324d3879a2bd918fc6a719fa/litellm/types/llms/openai.py#L392)
|
[**See All Message Values**](https://github.com/BerriAI/litellm/blob/main/litellm/types/llms/openai.py#L664)
|
||||||
|
|
||||||
|
#### Content Types
|
||||||
|
|
||||||
|
`content` can be a string (text only) or a list of content blocks (multimodal):
|
||||||
|
|
||||||
|
| Type | Description | Docs |
|
||||||
|
|------|-------------|------|
|
||||||
|
| `text` | Text content | [Type Definition](https://github.com/BerriAI/litellm/blob/main/litellm/types/llms/openai.py#L598) |
|
||||||
|
| `image_url` | Images | [Vision](./vision.md) |
|
||||||
|
| `input_audio` | Audio input | [Audio](./audio.md) |
|
||||||
|
| `video_url` | Video input | [Type Definition](https://github.com/BerriAI/litellm/blob/main/litellm/types/llms/openai.py#L625) |
|
||||||
|
| `file` | Files | [Document Understanding](./document_understanding.md) |
|
||||||
|
| `document` | Documents/PDFs | [Document Understanding](./document_understanding.md) |
|
||||||
|
|
||||||
|
**Examples:**
|
||||||
|
```python
|
||||||
|
# Text
|
||||||
|
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello!"}]}]
|
||||||
|
|
||||||
|
# Image
|
||||||
|
messages=[{"role": "user", "content": [{"type": "image_url", "image_url": {"url": "https://example.com/image.jpg"}}]}]
|
||||||
|
|
||||||
|
# Audio
|
||||||
|
messages=[{"role": "user", "content": [{"type": "input_audio", "input_audio": {"data": "<base64>", "format": "wav"}}]}]
|
||||||
|
|
||||||
|
# Video
|
||||||
|
messages=[{"role": "user", "content": [{"type": "video_url", "video_url": {"url": "https://example.com/video.mp4"}}]}]
|
||||||
|
|
||||||
|
# File
|
||||||
|
messages=[{"role": "user", "content": [{"type": "file", "file": {"file_id": "https://example.com/doc.pdf"}}]}]
|
||||||
|
|
||||||
|
# Document
|
||||||
|
messages=[{"role": "user", "content": [{"type": "document", "source": {"type": "text", "media_type": "application/pdf", "data": "<base64>"}}]}]
|
||||||
|
|
||||||
|
# Combining multiple types (multimodal)
|
||||||
|
messages=[{"role": "user", "content": [
|
||||||
|
{"type": "text", "text": "Generate a product description based on this image"},
|
||||||
|
{"type": "image_url", "image_url": {"url": "https://example.com/image.jpg"}}
|
||||||
|
]}]
|
||||||
|
```
|
||||||
|
|
||||||
## Optional Fields
|
## Optional Fields
|
||||||
|
|
||||||
|
|
@ -159,6 +199,8 @@ def completion(
|
||||||
- `include_usage` *boolean (optional)* - If set, an additional chunk will be streamed before the data: [DONE] message. The usage field on this chunk shows the token usage statistics for the entire request, and the choices field will always be an empty array. All other chunks will also include a usage field, but with a null value.
|
- `include_usage` *boolean (optional)* - If set, an additional chunk will be streamed before the data: [DONE] message. The usage field on this chunk shows the token usage statistics for the entire request, and the choices field will always be an empty array. All other chunks will also include a usage field, but with a null value.
|
||||||
|
|
||||||
- `stop`: *string/ array/ null (optional)* - Up to 4 sequences where the API will stop generating further tokens.
|
- `stop`: *string/ array/ null (optional)* - Up to 4 sequences where the API will stop generating further tokens.
|
||||||
|
|
||||||
|
**Note**: OpenAI supports a maximum of 4 stop sequences. If you provide more than 4, LiteLLM will automatically truncate the list to the first 4 elements. To disable this automatic truncation, set `litellm.disable_stop_sequence_limit = True`.
|
||||||
|
|
||||||
- `max_completion_tokens`: *integer (optional)* - An upper bound for the number of tokens that can be generated for a completion, including visible output tokens and reasoning tokens.
|
- `max_completion_tokens`: *integer (optional)* - An upper bound for the number of tokens that can be generated for a completion, including visible output tokens and reasoning tokens.
|
||||||
|
|
||||||
|
|
@ -174,11 +216,11 @@ def completion(
|
||||||
|
|
||||||
- `seed`: *integer or null (optional)* - This feature is in Beta. If specified, our system will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result. Determinism is not guaranteed, and you should refer to the `system_fingerprint` response parameter to monitor changes in the backend.
|
- `seed`: *integer or null (optional)* - This feature is in Beta. If specified, our system will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result. Determinism is not guaranteed, and you should refer to the `system_fingerprint` response parameter to monitor changes in the backend.
|
||||||
|
|
||||||
- `tools`: *array (optional)* - A list of tools the model may call. Currently, only functions are supported as a tool. Use this to provide a list of functions the model may generate JSON inputs for.
|
- `tools`: *array (optional)* - A list of tools the model may call. Use this to provide a list of functions the model may generate JSON inputs for.
|
||||||
|
|
||||||
- `type`: *string* - The type of the tool. Currently, only function is supported.
|
- `type`: *string* - The type of the tool. You can set this to `"function"` or `"mcp"` (matching the `/responses` schema) to call LiteLLM-registered MCP servers directly from `/chat/completions`.
|
||||||
|
|
||||||
- `function`: *object* - Required.
|
- `function`: *object* - Required for function tools.
|
||||||
|
|
||||||
- `tool_choice`: *string or object (optional)* - Controls which (if any) function is called by the model. none means the model will not call a function and instead generates a message. auto means the model can pick between generating a message or calling a function. Specifying a particular function via `{"type": "function", "function": {"name": "my_function"}}` forces the model to call that function.
|
- `tool_choice`: *string or object (optional)* - Controls which (if any) function is called by the model. none means the model will not call a function and instead generates a message. auto means the model can pick between generating a message or calling a function. Specifying a particular function via `{"type": "function", "function": {"name": "my_function"}}` forces the model to call that function.
|
||||||
|
|
||||||
|
|
@ -247,4 +289,3 @@ def completion(
|
||||||
- `eos_token`: *string (optional)* - Initial string applied at the end of a sequence
|
- `eos_token`: *string (optional)* - Initial string applied at the end of a sequence
|
||||||
|
|
||||||
- `hf_model_name`: *string (optional)* - [Sagemaker Only] The corresponding huggingface name of the model, used to pull the right chat template for the model.
|
- `hf_model_name`: *string (optional)* - [Sagemaker Only] The corresponding huggingface name of the model, used to pull the right chat template for the model.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -341,4 +341,90 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
||||||
```
|
```
|
||||||
|
|
||||||
</TabItem>
|
</TabItem>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
|
## Gemini - Native JSON Schema Format (Gemini 2.0+)
|
||||||
|
|
||||||
|
Gemini 2.0+ models automatically use the native `responseJsonSchema` parameter, which provides better compatibility with standard JSON Schema format.
|
||||||
|
|
||||||
|
### Benefits (Gemini 2.0+):
|
||||||
|
- Standard JSON Schema format (lowercase types like `string`, `object`)
|
||||||
|
- Supports `additionalProperties: false` for stricter validation
|
||||||
|
- Better compatibility with Pydantic's `model_json_schema()`
|
||||||
|
- No `propertyOrdering` required
|
||||||
|
|
||||||
|
### Usage
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="sdk" label="SDK">
|
||||||
|
|
||||||
|
```python
|
||||||
|
from litellm import completion
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
class UserInfo(BaseModel):
|
||||||
|
name: str
|
||||||
|
age: int
|
||||||
|
|
||||||
|
response = completion(
|
||||||
|
model="gemini/gemini-2.0-flash",
|
||||||
|
messages=[{"role": "user", "content": "Extract: John is 25 years old"}],
|
||||||
|
response_format={
|
||||||
|
"type": "json_schema",
|
||||||
|
"json_schema": {
|
||||||
|
"name": "user_info",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"age": {"type": "integer"}
|
||||||
|
},
|
||||||
|
"required": ["name", "age"],
|
||||||
|
"additionalProperties": False # Supported on Gemini 2.0+
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="proxy" label="PROXY">
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||||
|
-d '{
|
||||||
|
"model": "gemini-2.0-flash",
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Extract: John is 25 years old"}
|
||||||
|
],
|
||||||
|
"response_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"json_schema": {
|
||||||
|
"name": "user_info",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"name": {"type": "string"},
|
||||||
|
"age": {"type": "integer"}
|
||||||
|
},
|
||||||
|
"required": ["name", "age"],
|
||||||
|
"additionalProperties": false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
|
### Model Behavior
|
||||||
|
|
||||||
|
| Model | Format Used | `additionalProperties` Support |
|
||||||
|
|-------|-------------|-------------------------------|
|
||||||
|
| Gemini 2.0+ | `responseJsonSchema` (JSON Schema) | ✅ Yes |
|
||||||
|
| Gemini 1.5 | `responseSchema` (OpenAPI) | ❌ No |
|
||||||
|
|
||||||
|
LiteLLM automatically selects the appropriate format based on the model version.
|
||||||
|
|
@ -100,7 +100,7 @@ from litellm import cost_per_token
|
||||||
|
|
||||||
prompt_tokens = 5
|
prompt_tokens = 5
|
||||||
completion_tokens = 10
|
completion_tokens = 10
|
||||||
prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-3.5-turbo", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens))
|
prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar = cost_per_token(model="gpt-3.5-turbo", prompt_tokens=prompt_tokens, completion_tokens=completion_tokens)
|
||||||
|
|
||||||
print(prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar)
|
print(prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar)
|
||||||
```
|
```
|
||||||
|
|
@ -162,7 +162,7 @@ print(model_cost) # {'gpt-3.5-turbo': {'max_tokens': 4000, 'input_cost_per_token
|
||||||
|
|
||||||
**Dictionary**
|
**Dictionary**
|
||||||
```python
|
```python
|
||||||
from litellm import register_model
|
import litellm
|
||||||
|
|
||||||
litellm.register_model({
|
litellm.register_model({
|
||||||
"gpt-4": {
|
"gpt-4": {
|
||||||
|
|
|
||||||
|
|
@ -18,16 +18,46 @@ Each provider uses their own search backend:
|
||||||
|
|
||||||
| Provider | Search Engine | Notes |
|
| Provider | Search Engine | Notes |
|
||||||
|----------|---------------|-------|
|
|----------|---------------|-------|
|
||||||
| **OpenAI** (`gpt-4o-search-preview`) | OpenAI's internal search | Real-time web data |
|
| **OpenAI** (`gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview`) | OpenAI's internal search | Real-time web data |
|
||||||
| **xAI** (`grok-3`) | xAI's search + X/Twitter | Real-time social media data |
|
| **xAI** (`grok-3`) | xAI's search + X/Twitter | Real-time social media data |
|
||||||
| **Google AI/Vertex** (`gemini-2.0-flash`) | **Google Search** | Uses actual Google search results |
|
| **Google AI/Vertex** (`gemini-2.0-flash`) | **Google Search** | Uses actual Google search results |
|
||||||
| **Anthropic** (`claude-3-5-sonnet`) | Anthropic's web search | Real-time web data |
|
| **Anthropic** (`claude-3-5-sonnet`) | Anthropic's web search | Real-time web data |
|
||||||
| **Perplexity** | Perplexity's search engine | AI-powered search and reasoning |
|
| **Perplexity** | Perplexity's search engine | AI-powered search and reasoning |
|
||||||
|
|
||||||
|
:::warning Important: Only Search Models Support `web_search_options`
|
||||||
|
For OpenAI, only dedicated search models support the `web_search_options` parameter:
|
||||||
|
- `gpt-4o-search-preview`
|
||||||
|
- `gpt-4o-mini-search-preview`
|
||||||
|
- `gpt-5-search-api`
|
||||||
|
|
||||||
|
**Regular models like `gpt-5`, `gpt-4.1`, `gpt-4o` do not support `web_search_options`**
|
||||||
|
:::
|
||||||
|
|
||||||
|
:::tip The `web_search_options` parameter is optional
|
||||||
|
Search models (like `gpt-4o-search-preview`) **automatically search the web** even without the `web_search_options` parameter.
|
||||||
|
|
||||||
|
Use `web_search_options` when you need to:
|
||||||
|
- Adjust `search_context_size` (`"low"`, `"medium"`, `"high"`)
|
||||||
|
- Specify `user_location` for localized results
|
||||||
|
:::
|
||||||
|
|
||||||
:::info
|
:::info
|
||||||
**Anthropic Web Search Models**: Claude models that support web search: `claude-3-5-sonnet-latest`, `claude-3-5-sonnet-20241022`, `claude-3-5-haiku-latest`, `claude-3-5-haiku-20241022`, `claude-3-7-sonnet-20250219`
|
**Anthropic Web Search Models**: Claude models that support web search: `claude-3-5-sonnet-latest`, `claude-3-5-sonnet-20241022`, `claude-3-5-haiku-latest`, `claude-3-5-haiku-20241022`, `claude-3-7-sonnet-20250219`
|
||||||
:::
|
:::
|
||||||
|
|
||||||
|
## OpenAI Web Search: Two Approaches
|
||||||
|
|
||||||
|
OpenAI offers two distinct ways to use web search depending on the endpoint and model:
|
||||||
|
|
||||||
|
| Approach | Endpoint | Models | How to enable |
|
||||||
|
|----------|----------|--------|---------------|
|
||||||
|
| **Search Models** | `/chat/completions` | `gpt-5-search-api`, `gpt-4o-search-preview`, `gpt-4o-mini-search-preview` | Pass `web_search_options` parameter |
|
||||||
|
| **Web Search Tool** | `/responses` | `gpt-5`, `gpt-4.1`, `gpt-4o`, and other regular models | Pass `web_search_preview` tool |
|
||||||
|
|
||||||
|
:::tip Search models search automatically
|
||||||
|
Search models like `gpt-5-search-api` **automatically search the web** even without the `web_search_options` parameter. Use `web_search_options` to set `search_context_size` (`"low"`, `"medium"`, `"high"`) or specify `user_location` for localized results.
|
||||||
|
:::
|
||||||
|
|
||||||
## `/chat/completions` (litellm.completion)
|
## `/chat/completions` (litellm.completion)
|
||||||
|
|
||||||
### Quick Start
|
### Quick Start
|
||||||
|
|
@ -39,7 +69,7 @@ Each provider uses their own search backend:
|
||||||
from litellm import completion
|
from litellm import completion
|
||||||
|
|
||||||
response = completion(
|
response = completion(
|
||||||
model="openai/gpt-4o-search-preview",
|
model="openai/gpt-5-search-api",
|
||||||
messages=[
|
messages=[
|
||||||
{
|
{
|
||||||
"role": "user",
|
"role": "user",
|
||||||
|
|
@ -59,31 +89,36 @@ response = completion(
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
model_list:
|
model_list:
|
||||||
# OpenAI
|
# OpenAI search models
|
||||||
|
- model_name: gpt-5-search-api
|
||||||
|
litellm_params:
|
||||||
|
model: openai/gpt-5-search-api
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
|
||||||
- model_name: gpt-4o-search-preview
|
- model_name: gpt-4o-search-preview
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: openai/gpt-4o-search-preview
|
model: openai/gpt-4o-search-preview
|
||||||
api_key: os.environ/OPENAI_API_KEY
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
|
||||||
# xAI
|
# xAI
|
||||||
- model_name: grok-3
|
- model_name: grok-3
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: xai/grok-3
|
model: xai/grok-3
|
||||||
api_key: os.environ/XAI_API_KEY
|
api_key: os.environ/XAI_API_KEY
|
||||||
|
|
||||||
# Anthropic
|
# Anthropic
|
||||||
- model_name: claude-3-5-sonnet-latest
|
- model_name: claude-3-5-sonnet-latest
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: anthropic/claude-3-5-sonnet-latest
|
model: anthropic/claude-3-5-sonnet-latest
|
||||||
api_key: os.environ/ANTHROPIC_API_KEY
|
api_key: os.environ/ANTHROPIC_API_KEY
|
||||||
|
|
||||||
# VertexAI
|
# VertexAI
|
||||||
- model_name: gemini-2-flash
|
- model_name: gemini-2-flash
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: gemini-2.0-flash
|
model: gemini-2.0-flash
|
||||||
vertex_project: your-project-id
|
vertex_project: your-project-id
|
||||||
vertex_location: us-central1
|
vertex_location: us-central1
|
||||||
|
|
||||||
# Google AI Studio
|
# Google AI Studio
|
||||||
- model_name: gemini-2-flash-studio
|
- model_name: gemini-2-flash-studio
|
||||||
litellm_params:
|
litellm_params:
|
||||||
|
|
@ -91,13 +126,13 @@ model_list:
|
||||||
api_key: os.environ/GOOGLE_API_KEY
|
api_key: os.environ/GOOGLE_API_KEY
|
||||||
```
|
```
|
||||||
|
|
||||||
2. Start the proxy
|
2. Start the proxy
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
litellm --config /path/to/config.yaml
|
litellm --config /path/to/config.yaml
|
||||||
```
|
```
|
||||||
|
|
||||||
3. Test it!
|
3. Test it!
|
||||||
|
|
||||||
```python showLineNumbers
|
```python showLineNumbers
|
||||||
from openai import OpenAI
|
from openai import OpenAI
|
||||||
|
|
@ -109,13 +144,18 @@ client = OpenAI(
|
||||||
)
|
)
|
||||||
|
|
||||||
response = client.chat.completions.create(
|
response = client.chat.completions.create(
|
||||||
model="grok-3", # or any other web search enabled model
|
model="gpt-5-search-api", # or any other web search enabled model
|
||||||
messages=[
|
messages=[
|
||||||
{
|
{
|
||||||
"role": "user",
|
"role": "user",
|
||||||
"content": "What was a positive news story from today?"
|
"content": "What was a positive news story from today?"
|
||||||
}
|
}
|
||||||
]
|
],
|
||||||
|
extra_body={
|
||||||
|
"web_search_options": {
|
||||||
|
"search_context_size": "medium"
|
||||||
|
}
|
||||||
|
}
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
</TabItem>
|
</TabItem>
|
||||||
|
|
@ -132,7 +172,7 @@ from litellm import completion
|
||||||
|
|
||||||
# Customize search context size
|
# Customize search context size
|
||||||
response = completion(
|
response = completion(
|
||||||
model="openai/gpt-4o-search-preview",
|
model="openai/gpt-5-search-api",
|
||||||
messages=[
|
messages=[
|
||||||
{
|
{
|
||||||
"role": "user",
|
"role": "user",
|
||||||
|
|
@ -240,6 +280,12 @@ response = client.chat.completions.create(
|
||||||
|
|
||||||
## `/responses` (litellm.responses)
|
## `/responses` (litellm.responses)
|
||||||
|
|
||||||
|
Use the `web_search_preview` tool with models like `gpt-5`, `gpt-4.1`, `gpt-4o`, etc.
|
||||||
|
|
||||||
|
:::info
|
||||||
|
Search-dedicated models like `gpt-5-search-api` and `gpt-4o-search-preview` do **not** support the `/responses` endpoint. Use them with `/chat/completions` + `web_search_options` instead (see above).
|
||||||
|
:::
|
||||||
|
|
||||||
### Quick Start
|
### Quick Start
|
||||||
|
|
||||||
<Tabs>
|
<Tabs>
|
||||||
|
|
@ -249,18 +295,14 @@ response = client.chat.completions.create(
|
||||||
from litellm import responses
|
from litellm import responses
|
||||||
|
|
||||||
response = responses(
|
response = responses(
|
||||||
model="openai/gpt-4o",
|
model="openai/gpt-5",
|
||||||
input=[
|
input="What is the capital of France?",
|
||||||
{
|
|
||||||
"role": "user",
|
|
||||||
"content": "What was a positive news story from today?"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
tools=[{
|
tools=[{
|
||||||
"type": "web_search_preview" # enables web search with default medium context size
|
"type": "web_search_preview" # enables web search with default medium context size
|
||||||
}]
|
}]
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
</TabItem>
|
</TabItem>
|
||||||
<TabItem value="proxy" label="PROXY">
|
<TabItem value="proxy" label="PROXY">
|
||||||
|
|
||||||
|
|
@ -268,19 +310,24 @@ response = responses(
|
||||||
|
|
||||||
```yaml
|
```yaml
|
||||||
model_list:
|
model_list:
|
||||||
- model_name: gpt-4o
|
- model_name: gpt-5
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: openai/gpt-4o
|
model: openai/gpt-5
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
|
||||||
|
- model_name: gpt-4.1
|
||||||
|
litellm_params:
|
||||||
|
model: openai/gpt-4.1
|
||||||
api_key: os.environ/OPENAI_API_KEY
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
```
|
```
|
||||||
|
|
||||||
2. Start the proxy
|
2. Start the proxy
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
litellm --config /path/to/config.yaml
|
litellm --config /path/to/config.yaml
|
||||||
```
|
```
|
||||||
|
|
||||||
3. Test it!
|
3. Test it!
|
||||||
|
|
||||||
```python showLineNumbers
|
```python showLineNumbers
|
||||||
from openai import OpenAI
|
from openai import OpenAI
|
||||||
|
|
@ -292,11 +339,11 @@ client = OpenAI(
|
||||||
)
|
)
|
||||||
|
|
||||||
response = client.responses.create(
|
response = client.responses.create(
|
||||||
model="gpt-4o",
|
model="gpt-5",
|
||||||
tools=[{
|
tools=[{
|
||||||
"type": "web_search_preview"
|
"type": "web_search_preview"
|
||||||
}],
|
}],
|
||||||
input="What was a positive news story from today?",
|
input="What is the capital of France?",
|
||||||
)
|
)
|
||||||
|
|
||||||
print(response.output_text)
|
print(response.output_text)
|
||||||
|
|
@ -314,13 +361,8 @@ from litellm import responses
|
||||||
|
|
||||||
# Customize search context size
|
# Customize search context size
|
||||||
response = responses(
|
response = responses(
|
||||||
model="openai/gpt-4o",
|
model="openai/gpt-5",
|
||||||
input=[
|
input="What is the capital of France?",
|
||||||
{
|
|
||||||
"role": "user",
|
|
||||||
"content": "What was a positive news story from today?"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
tools=[{
|
tools=[{
|
||||||
"type": "web_search_preview",
|
"type": "web_search_preview",
|
||||||
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
||||||
|
|
@ -341,12 +383,12 @@ client = OpenAI(
|
||||||
|
|
||||||
# Customize search context size
|
# Customize search context size
|
||||||
response = client.responses.create(
|
response = client.responses.create(
|
||||||
model="gpt-4o",
|
model="gpt-5",
|
||||||
tools=[{
|
tools=[{
|
||||||
"type": "web_search_preview",
|
"type": "web_search_preview",
|
||||||
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
||||||
}],
|
}],
|
||||||
input="What was a positive news story from today?",
|
input="What is the capital of France?",
|
||||||
)
|
)
|
||||||
|
|
||||||
print(response.output_text)
|
print(response.output_text)
|
||||||
|
|
@ -400,14 +442,14 @@ model_list:
|
||||||
web_search_options:
|
web_search_options:
|
||||||
search_context_size: "high" # Options: "low", "medium", "high"
|
search_context_size: "high" # Options: "low", "medium", "high"
|
||||||
|
|
||||||
# Different context size for different models
|
# OpenAI search model with custom context size
|
||||||
- model_name: gpt-4o-search-preview
|
- model_name: gpt-5-search-api
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: openai/gpt-4o-search-preview
|
model: openai/gpt-5-search-api
|
||||||
api_key: os.environ/OPENAI_API_KEY
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
web_search_options:
|
web_search_options:
|
||||||
search_context_size: "low"
|
search_context_size: "low"
|
||||||
|
|
||||||
# Gemini with medium context (default)
|
# Gemini with medium context (default)
|
||||||
- model_name: gemini-2-flash
|
- model_name: gemini-2-flash
|
||||||
litellm_params:
|
litellm_params:
|
||||||
|
|
@ -432,6 +474,7 @@ Use `litellm.supports_web_search(model="model_name")` -> returns `True` if model
|
||||||
|
|
||||||
```python showLineNumbers
|
```python showLineNumbers
|
||||||
# Check OpenAI models
|
# Check OpenAI models
|
||||||
|
assert litellm.supports_web_search(model="openai/gpt-5-search-api") == True
|
||||||
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
|
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
|
||||||
|
|
||||||
# Check xAI models
|
# Check xAI models
|
||||||
|
|
@ -455,13 +498,20 @@ assert litellm.supports_web_search(model="gemini/gemini-2.0-flash") == True
|
||||||
```yaml
|
```yaml
|
||||||
model_list:
|
model_list:
|
||||||
# OpenAI
|
# OpenAI
|
||||||
|
- model_name: gpt-5-search-api
|
||||||
|
litellm_params:
|
||||||
|
model: openai/gpt-5-search-api
|
||||||
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
|
model_info:
|
||||||
|
supports_web_search: True
|
||||||
|
|
||||||
- model_name: gpt-4o-search-preview
|
- model_name: gpt-4o-search-preview
|
||||||
litellm_params:
|
litellm_params:
|
||||||
model: openai/gpt-4o-search-preview
|
model: openai/gpt-4o-search-preview
|
||||||
api_key: os.environ/OPENAI_API_KEY
|
api_key: os.environ/OPENAI_API_KEY
|
||||||
model_info:
|
model_info:
|
||||||
supports_web_search: True
|
supports_web_search: True
|
||||||
|
|
||||||
# xAI
|
# xAI
|
||||||
- model_name: grok-3
|
- model_name: grok-3
|
||||||
litellm_params:
|
litellm_params:
|
||||||
|
|
@ -516,6 +566,12 @@ Expected Response
|
||||||
```json showLineNumbers
|
```json showLineNumbers
|
||||||
{
|
{
|
||||||
"data": [
|
"data": [
|
||||||
|
{
|
||||||
|
"model_group": "gpt-5-search-api",
|
||||||
|
"providers": ["openai"],
|
||||||
|
"max_tokens": 128000,
|
||||||
|
"supports_web_search": true
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"model_group": "gpt-4o-search-preview",
|
"model_group": "gpt-4o-search-preview",
|
||||||
"providers": ["openai"],
|
"providers": ["openai"],
|
||||||
|
|
|
||||||
|
|
@ -21,6 +21,7 @@ Looking for how to use Code Interpreter? See the [Code Interpreter Guide](/docs/
|
||||||
|
|
||||||
| Endpoint | Method | Description |
|
| Endpoint | Method | Description |
|
||||||
|----------|--------|-------------|
|
|----------|--------|-------------|
|
||||||
|
| `/v1/containers/{container_id}/files` | POST | Upload file to container |
|
||||||
| `/v1/containers/{container_id}/files` | GET | List files in container |
|
| `/v1/containers/{container_id}/files` | GET | List files in container |
|
||||||
| `/v1/containers/{container_id}/files/{file_id}` | GET | Get file metadata |
|
| `/v1/containers/{container_id}/files/{file_id}` | GET | Get file metadata |
|
||||||
| `/v1/containers/{container_id}/files/{file_id}/content` | GET | Download file content |
|
| `/v1/containers/{container_id}/files/{file_id}/content` | GET | Download file content |
|
||||||
|
|
@ -28,6 +29,45 @@ Looking for how to use Code Interpreter? See the [Code Interpreter Guide](/docs/
|
||||||
|
|
||||||
## LiteLLM Python SDK
|
## LiteLLM Python SDK
|
||||||
|
|
||||||
|
### Upload Container File
|
||||||
|
|
||||||
|
Upload files directly to a container session. This is useful when `/chat/completions` or `/responses` sends files to the container but the input file type is limited to PDF. This endpoint lets you work with other file types like CSV, Excel, Python scripts, etc.
|
||||||
|
|
||||||
|
```python showLineNumbers title="upload_container_file.py"
|
||||||
|
from litellm import upload_container_file
|
||||||
|
|
||||||
|
# Upload a CSV file
|
||||||
|
file = upload_container_file(
|
||||||
|
container_id="cntr_123...",
|
||||||
|
file=("data.csv", open("data.csv", "rb").read(), "text/csv"),
|
||||||
|
custom_llm_provider="openai"
|
||||||
|
)
|
||||||
|
|
||||||
|
print(f"Uploaded: {file.id}")
|
||||||
|
print(f"Path: {file.path}")
|
||||||
|
```
|
||||||
|
|
||||||
|
**Async:**
|
||||||
|
|
||||||
|
```python showLineNumbers title="aupload_container_file.py"
|
||||||
|
from litellm import aupload_container_file
|
||||||
|
|
||||||
|
file = await aupload_container_file(
|
||||||
|
container_id="cntr_123...",
|
||||||
|
file=("script.py", b"print('hello world')", "text/x-python"),
|
||||||
|
custom_llm_provider="openai"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
**Supported file formats:**
|
||||||
|
- CSV (`.csv`)
|
||||||
|
- Excel (`.xlsx`)
|
||||||
|
- Python scripts (`.py`)
|
||||||
|
- JSON (`.json`)
|
||||||
|
- Markdown (`.md`)
|
||||||
|
- Text files (`.txt`)
|
||||||
|
- And more...
|
||||||
|
|
||||||
### List Container Files
|
### List Container Files
|
||||||
|
|
||||||
```python showLineNumbers title="list_container_files.py"
|
```python showLineNumbers title="list_container_files.py"
|
||||||
|
|
@ -103,6 +143,40 @@ print(f"Deleted: {result.deleted}")
|
||||||
import Tabs from '@theme/Tabs';
|
import Tabs from '@theme/Tabs';
|
||||||
import TabItem from '@theme/TabItem';
|
import TabItem from '@theme/TabItem';
|
||||||
|
|
||||||
|
### Upload File
|
||||||
|
|
||||||
|
<Tabs>
|
||||||
|
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||||
|
|
||||||
|
```python showLineNumbers title="upload_file.py"
|
||||||
|
from openai import OpenAI
|
||||||
|
|
||||||
|
client = OpenAI(
|
||||||
|
api_key="sk-1234",
|
||||||
|
base_url="http://localhost:4000"
|
||||||
|
)
|
||||||
|
|
||||||
|
file = client.containers.files.create(
|
||||||
|
container_id="cntr_123...",
|
||||||
|
file=open("data.csv", "rb")
|
||||||
|
)
|
||||||
|
|
||||||
|
print(f"Uploaded: {file.id}")
|
||||||
|
print(f"Path: {file.path}")
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
<TabItem value="curl" label="curl">
|
||||||
|
|
||||||
|
```bash showLineNumbers title="upload_file.sh"
|
||||||
|
curl "http://localhost:4000/v1/containers/cntr_123.../files" \
|
||||||
|
-H "Authorization: Bearer sk-1234" \
|
||||||
|
-F file="@data.csv"
|
||||||
|
```
|
||||||
|
|
||||||
|
</TabItem>
|
||||||
|
</Tabs>
|
||||||
|
|
||||||
### List Files
|
### List Files
|
||||||
|
|
||||||
<Tabs>
|
<Tabs>
|
||||||
|
|
@ -236,6 +310,13 @@ curl -X DELETE "http://localhost:4000/v1/containers/cntr_123.../files/cfile_456.
|
||||||
|
|
||||||
## Parameters
|
## Parameters
|
||||||
|
|
||||||
|
### Upload File
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Description |
|
||||||
|
|-----------|------|----------|-------------|
|
||||||
|
| `container_id` | string | Yes | Container ID |
|
||||||
|
| `file` | FileTypes | Yes | File to upload. Can be a tuple of (filename, content, content_type), file-like object, or bytes |
|
||||||
|
|
||||||
### List Files
|
### List Files
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
| Parameter | Type | Required | Description |
|
||||||
|
|
|
||||||
|
|
@ -1,45 +1,100 @@
|
||||||
# Contributing - UI
|
# Contributing - UI
|
||||||
|
|
||||||
Here's how to run the LiteLLM UI locally for making changes:
|
Thanks for contributing to the LiteLLM UI! This guide will help you set up your local development environment.
|
||||||
|
|
||||||
|
|
||||||
|
## 1. Clone the repo
|
||||||
|
|
||||||
## 1. Clone the repo
|
|
||||||
```bash
|
```bash
|
||||||
git clone https://github.com/BerriAI/litellm.git
|
git clone https://github.com/BerriAI/litellm.git
|
||||||
|
cd litellm
|
||||||
```
|
```
|
||||||
|
|
||||||
## 2. Start the UI + Proxy
|
## 2. Start the Proxy
|
||||||
|
|
||||||
**2.1 Start the proxy on port 4000**
|
Create a config file (e.g., `config.yaml`):
|
||||||
|
|
||||||
Tell the proxy where the UI is located
|
```yaml
|
||||||
```bash
|
model_list:
|
||||||
DATABASE_URL = "postgresql://<user>:<password>@<host>:<port>/<dbname>"
|
- model_name: gpt-4o
|
||||||
LITELLM_MASTER_KEY = "sk-1234"
|
litellm_params:
|
||||||
STORE_MODEL_IN_DB = "True"
|
model: openai/gpt-4o
|
||||||
|
|
||||||
|
general_settings:
|
||||||
|
master_key: sk-1234
|
||||||
|
database_url: postgresql://<user>:<password>@<host>:<port>/<dbname>
|
||||||
|
store_model_in_db: true
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Start the proxy on port 4000:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cd litellm/litellm/proxy
|
poetry run litellm --config config.yaml --port 4000
|
||||||
python3 proxy_cli.py --config /path/to/config.yaml --port 4000
|
|
||||||
```
|
```
|
||||||
|
|
||||||
**2.2 Start the UI**
|
The UI comes pre-built in the repo. Access it at `http://localhost:4000/ui`
|
||||||
|
|
||||||
Set the mode as development (this will assume the proxy is running on localhost:4000)
|
## 3. UI Development
|
||||||
```bash
|
|
||||||
npm install # install dependencies
|
There are two options for UI development:
|
||||||
```
|
|
||||||
|
### Option A: Development Mode (Hot Reload)
|
||||||
|
|
||||||
|
This runs the UI on port 3000 with hot reload. The proxy runs on port 4000.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cd litellm/ui/litellm-dashboard
|
cd ui/litellm-dashboard
|
||||||
|
npm install
|
||||||
npm run dev
|
npm run dev
|
||||||
|
|
||||||
# starts on http://0.0.0.0:3000
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## 3. Go to local UI
|
**Login flow:**
|
||||||
|
1. Go to `http://localhost:3000`
|
||||||
|
2. You'll be redirected to `http://localhost:4000/ui` for login
|
||||||
|
3. After logging in, manually navigate back to `http://localhost:3000/`
|
||||||
|
4. You're now authenticated and can develop with hot reload
|
||||||
|
|
||||||
|
:::note
|
||||||
|
If you experience redirect loops or authentication issues, clear your browser cookies for localhost or use Build Mode instead.
|
||||||
|
:::
|
||||||
|
|
||||||
|
### Option B: Build Mode
|
||||||
|
|
||||||
|
This builds the UI and copies it to the proxy. Changes require rebuilding.
|
||||||
|
|
||||||
|
1. Make your code changes in `ui/litellm-dashboard/src/`
|
||||||
|
|
||||||
|
2. Build the UI
|
||||||
|
```bash
|
||||||
|
cd ui/litellm-dashboard
|
||||||
|
npm install
|
||||||
|
npm run build
|
||||||
|
```
|
||||||
|
|
||||||
|
After building, copy the output to the proxy:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
http://0.0.0.0:3000
|
cp -r out/* ../../litellm/proxy/_experimental/out/
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Then restart the proxy and access the UI at `http://localhost:4000/ui`
|
||||||
|
|
||||||
|
## 4. Submitting a PR
|
||||||
|
|
||||||
|
1. Create a new branch for your changes:
|
||||||
|
```bash
|
||||||
|
git checkout -b feat/your-feature-name
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Stage and commit your changes:
|
||||||
|
```bash
|
||||||
|
git add .
|
||||||
|
git commit -m "feat: description of your changes"
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Push to your fork:
|
||||||
|
```bash
|
||||||
|
git push origin feat/your-feature-name
|
||||||
|
```
|
||||||
|
|
||||||
|
4. Create a Pull Request on GitHub following the [PR template](https://github.com/BerriAI/litellm/blob/main/.github/pull_request_template.md)
|
||||||
|
|
|
||||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue